{"as_of":"2026-08-06T18:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a70aec205c1cb6069a7f5a13b3b7b6c1ca5ab818ca055d5d3951cb875b2d4eae","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-23T20:31:24.494060Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T15:21:28.809697Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-03T00:47:30.552861Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"cited_work":{"arxiv_id":"2409.18512","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.18512","snapshot_observed_at":"2026-07-03T00:47:30.552861Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","venue":"cs.SD","work_id":"7dbac3b9-0e06-4d5f-833c-7616bdedf69e","year":2024},"citing_paper":{"arxiv_id":"2604.26347","last_updated":"2026-07-22T04:43:04Z","snapshot_observed_at":"2026-08-02T15:21:28.271483Z","submitted_at":"2026-04-29T06:59:48Z","title":"The False Resonance: A Critical Examination of Emotion Embedding Similarity for Speech Generation Evaluation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-07T12:34:40.888089Z"},"links":{"cited_paper":"/paper/2409.18512","citing_paper":"/paper/2604.26347"},"observation_digest":"sha256:278d1de4aea13c2bee7f04fc2a34b3e19f871b913dc81b79ced6c20d7c8340eb","observation_id":"04253d91-b32c-472f-8ddd-81d871d9c2c4","resolution":{"observed_at":"2026-05-12T09:11:27.091266Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.18512","snapshot_observed_at":"2026-08-02T15:21:28.809697Z","title":"Emopro: A prompt selection strategy for emotional expression in lm-based speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.26347","last_updated":"2026-07-22T04:43:04Z","snapshot_observed_at":"2026-08-02T15:21:28.271483Z","submitted_at":"2026-04-29T06:59:48Z","title":"The False Resonance: A Critical Examination of Emotion Embedding Similarity for Speech Generation Evaluation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T15:21:28.809697Z"},"links":{"cited_paper":"/paper/2409.18512","citing_paper":"/paper/2604.26347"},"observation_digest":"sha256:7ca5a59823275672f3863990fe9bea7535cce9fcc6ef089f7ef220ce017287b8","observation_id":"02415a3e-7d8b-468f-a09f-c4751d9e55d9","resolution":{"observed_at":"2026-08-02T15:21:28.809697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"cited_work":{"arxiv_id":"2409.18512","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.18512","snapshot_observed_at":"2026-07-03T00:47:30.552861Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","venue":"cs.SD","work_id":"7dbac3b9-0e06-4d5f-833c-7616bdedf69e","year":2024},"citing_paper":{"arxiv_id":"2606.20650","last_updated":"2026-06-08T06:54:55Z","snapshot_observed_at":"2026-08-02T23:08:38.292740Z","submitted_at":"2026-06-08T06:54:55Z","title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T17:01:13.972071Z"},"links":{"cited_paper":"/paper/2409.18512","citing_paper":"/paper/2606.20650"},"observation_digest":"sha256:49a1a3b54490ae42d370dda9c5cabe02b1b45985ec8e2e701dea880baba82a6a","observation_id":"234b4815-02f5-4a2b-9e88-cd6e0ca0ddb4","resolution":{"observed_at":"2026-07-03T00:47:30.554212Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2409.18512/citation-record","integrity":"/paper/2409.18512/integrity","json":"/paper/2409.18512/citation-record.json","paper":"/paper/2409.18512"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improving language understanding by generative pre- training","venue":null,"work_id":"36cae410-33c4-40e6-8493-56ad996a2204","year":2018},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:8edf8e20a6d88db81d0ec6bebe994e35c6161c4cf534ce3a54eaf6a1e3c9e71f","observation_id":"ad99742c-3ffc-4e72-bef6-190a5430f71d","resolution":{"observed_at":"2026-05-23T20:35:48.960930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-02T20:06:32.256011Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":"2301.02111","doi":"10.48550/arxiv.2301.02111","metadata_source":"pith","pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","venue":"cs.CL","work_id":"de8fb688-dd63-4942-92f6-66d98d5b6db2","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:9330dd7443188d916810b54c852b6ea332b1357609ec8f5becfc394fbb3e3d99","observation_id":"30f9dfec-c267-459c-b604-d7a7e2cf491e","resolution":{"observed_at":"2026-05-23T20:33:25.623334Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Speak, read and prompt: High-fidelity text-to-speech with minimal supervision","venue":null,"work_id":"f8b11e0b-9ef5-476a-8c02-bb455b805e9f","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:9d7f0c29581287deecc673774011527dedcfa7bb59df0840de217ac2c2d25bbf","observation_id":"41a2b1dd-079b-45c7-b714-8916ac2a2a2d","resolution":{"observed_at":"2026-05-23T20:35:48.957044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","venue":null,"work_id":"9fb6c143-eb67-4971-9d8c-6086cadd0d98","year":2021},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:9a8d78b20e6a4c44a21bd1fcf198823348c0953ac146d27b55a7697b517701b6","observation_id":"037ffe26-1e20-43f8-b9e4-ea3e935ba457","resolution":{"observed_at":"2026-05-23T20:35:48.953253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learn- ing speech representation from contrastive token-acoustic pretraining","venue":null,"work_id":"2f70d6b8-1b64-4c1f-ae58-390f7253d043","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:cdb8fb63224c62b854afa4783be5a1c27fab2323d40d37d4fbcfe767b04a4bff","observation_id":"f01f4e79-4c5e-45b6-bed7-27dfaa3c9b4c","resolution":{"observed_at":"2026-05-23T20:35:48.956374Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.13438","last_updated":"2022-10-24T17:52:02Z","snapshot_observed_at":"2026-08-03T16:47:47.192907Z","submitted_at":"2022-10-24T17:52:02Z","title":"High Fidelity Neural Audio Compression","version":1},"cited_work":{"arxiv_id":"2210.13438","doi":"10.48550/arxiv.2210.13438","metadata_source":"pith","pith_arxiv_id":"2210.13438","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"High Fidelity Neural Audio Compression","venue":"eess.AS","work_id":"bc645d2d-e9f2-4cb8-9a6d-bd557bc7a258","year":2022},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2210.13438","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:bd06034cdf0e96c1b0e837a0d9d7c232ad637f91a98d850de1a1fcd39a65b02f","observation_id":"25331a38-e140-4b57-858c-6ff85959fe2c","resolution":{"observed_at":"2026-05-23T20:33:25.618036Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-19T16:22:25.658509+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T16:22:25.658509+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":"2407.05407","doi":"10.48550/arxiv.2407.05407","metadata_source":"pith","pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","venue":"cs.SD","work_id":"e5ad925a-4045-49b5-b301-208bcbf3eca8","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:2084d50891d591c0482bb4a9b7bf07babea174285d85b0a928875064cec844eb","observation_id":"ce977247-b194-45cc-aab6-fd1a9270f528","resolution":{"observed_at":"2026-05-23T20:33:25.628224Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.12925","last_updated":"2023-06-22T14:37:54Z","snapshot_observed_at":"2026-08-02T02:42:46.945082Z","submitted_at":"2023-06-22T14:37:54Z","title":"AudioPaLM: A Large Language Model That Can Speak and Listen","version":1},"cited_work":{"arxiv_id":"2306.12925","doi":"10.21437/interspeech.2019-1873","metadata_source":"pith","pith_arxiv_id":"2306.12925","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AudioPaLM: A Large Language Model That Can Speak and Listen","venue":"cs.CL","work_id":"a828d91a-9395-420e-825d-6858006f228e","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2306.12925","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:e77461a9c256d0bf0f59dd67136d73651a447d27d75faa583d914dc788d44b7e","observation_id":"845d86b3-8597-459f-b362-021156237d04","resolution":{"observed_at":"2026-05-23T20:33:25.633096Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"VioLA: Conditional language models for speech recognition, synthesis, and translation","venue":null,"work_id":"c852f5bb-bfdf-4a5b-9ef2-6ff9d2bda168","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:705fdc6b07af9adef57d3108eafcb626ece85316d3e26399b3344892c9f42b57","observation_id":"b5c32be0-0f97-46bb-b4c3-b1ecb7d43a1d","resolution":{"observed_at":"2026-05-23T20:35:48.949626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Large lan- guage models are zero-shot reasoners","venue":null,"work_id":"990df09e-3ae6-426f-8e1d-00b0ebda37df","year":2022},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:059b8b2bc4fa87c7710eec2a5edfbf7d07294043e79007d71c6725e3d0a8c3cb","observation_id":"f44803bf-8ca9-4ce2-99ac-e5a902bf5f66","resolution":{"observed_at":"2026-05-23T20:35:48.952837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling instruction-finetuned language models","venue":null,"work_id":"c4164686-327b-4dc0-8b91-2754deb6f057","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:2d407d8366bf1f1bf4523aa02a286b7586a87f9be84300123b4bbee3239170cf","observation_id":"8b51d302-f83c-4283-b395-0dec0540f6d8","resolution":{"observed_at":"2026-05-23T20:35:48.959598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Retrieval-based prompt se- lection for code-related few-shot learning","venue":null,"work_id":"b20e36a3-e9ec-4388-ba30-234fe5090886","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:c79ac68a8df75da7a4f3c9ed4de32d3c8ee303d37dc4ff001ac6a31cd1e10774","observation_id":"d5625e6c-f2a4-4c3b-8f7c-1c155660c07b","resolution":{"observed_at":"2026-05-23T20:35:48.935925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to prompt for continual learning","venue":null,"work_id":"7c75f551-f24e-409e-90b7-4d0379fedb2c","year":2022},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:0448195973931a955295bee9e445aa02efcf11e516bec70c112ef58b72e9e1eb","observation_id":"8e68ddf2-4fd1-445b-880d-4961b4cf0ce5","resolution":{"observed_at":"2026-05-23T20:35:48.934275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.12822","last_updated":"2024-02-27T14:49:43Z","snapshot_observed_at":"2026-07-06T14:55:34.311264Z","submitted_at":"2023-02-24T18:58:06Z","title":"Automatic Prompt Augmentation and Selection with Chain-of-Thought from Labeled Data","version":3},"cited_work":{"arxiv_id":"2302.12822","doi":"10.48550/arxiv.2302.12822","metadata_source":"arxiv_reference","pith_arxiv_id":"2302.12822","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Automatic prompt augmentation and selection with chain-of-thought from labeled data","venue":"arXiv (Cornell University)","work_id":"c5370897-33d8-41a1-93df-be91cf1d34e2","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2302.12822","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:85749b534c0acbfdbdfcb0a4b74f1af312d2ce4e082e33c12201526a0b15c4be","observation_id":"19f305a6-b778-453d-a025-e067509cd618","resolution":{"observed_at":"2026-05-23T20:33:25.596984Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Universal information extraction as unified semantic matching","venue":null,"work_id":"2434fad8-8232-4ce3-944f-78a97d3ca34b","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:47591f42f690f631d96429bc8afe504760253c2a969e914be30f5e83453960b1","observation_id":"d8ba8b3f-8e7a-445f-b252-fe83809d15d0","resolution":{"observed_at":"2026-05-23T20:35:48.931247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sentence-bert: Sentence embeddings using siamese bert-networks","venue":null,"work_id":"17e8d371-0041-4da3-a2fb-4e88d93803a5","year":2019},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:c0d999de067b65f498b289f90c17900e83d37e8db272c4a7f2ed6e5f77007e97","observation_id":"ed3b68eb-c4ac-4634-99ea-8eb79382650d","resolution":{"observed_at":"2026-05-23T20:35:48.928572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06406","last_updated":"2024-06-11T19:54:35Z","snapshot_observed_at":"2026-08-02T07:36:17.630745Z","submitted_at":"2024-06-10T15:58:42Z","title":"Controlling Emotion in Text-to-Speech with Natural Language Prompts","version":2},"cited_work":{"arxiv_id":"2406.06406","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.06406","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Controlling emotion in text-to-speech with natural language prompts","venue":null,"work_id":"15bb1dc3-4bc9-482a-8bb3-5e9db3ae9626","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2406.06406","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:63814b6c0af7eb7e52721004dd7ebe8ef8ac8b4c49a92fcca10d91a6cd941ff1","observation_id":"b83fb258-e51b-4a4c-8554-331dbcdcf692","resolution":{"observed_at":"2026-05-23T20:33:25.581696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"cited_work":{"arxiv_id":"2406.02430","doi":"10.48550/arxiv.2406.02430","metadata_source":"pith","pith_arxiv_id":"2406.02430","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","venue":"eess.AS","work_id":"6e88ee95-1133-4302-a142-cdf8f9456a8d","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2406.02430","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:722317ab6b7f0ae067a271e244693be5591cf50867566a2bec569f1cf621e890","observation_id":"041024c9-5f47-488e-9eb4-24126add3958","resolution":{"observed_at":"2026-05-23T20:33:25.586949Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18398","last_updated":"2025-02-18T21:39:25Z","snapshot_observed_at":"2026-07-06T18:06:49.190172Z","submitted_at":"2024-04-29T03:19:39Z","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":"2404.18398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.18398","snapshot_observed_at":"2026-07-04T19:20:06.506505Z","title":"Mm-tts: A unified framework for multimodal, prompt-induced emotional text-to- speech synthesis","venue":null,"work_id":"a2a66868-db3d-4c0c-a7f4-ea0feac17fe3","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2404.18398","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:bc6ddb43cae3d71d8a9a1d581612dfc6a2b797b8a05c0a79a8ddeaa799112ba8","observation_id":"3be1f78d-e8b2-4032-8a1d-270876184055","resolution":{"observed_at":"2026-05-23T20:33:25.607564Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zmm-tts: Zero-shot multilingual and multi- speaker speech synthesis conditioned on self-supervised discrete speech representations","venue":null,"work_id":"477389f7-e236-4e3f-bd9f-f342436575d5","year":2024},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:de78c029183aeb830c44345ce48141d6b45e9ba19986cad8486da31858031602","observation_id":"3d1c4d59-b651-4169-8a74-dc93eb7c4d49","resolution":{"observed_at":"2026-05-23T20:35:48.925502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dnsmos: A non-intrusive perceptual objective speech quality metric to evaluate noise suppressors","venue":null,"work_id":"8dd2dc97-24e4-494a-ab77-832f7c84be21","year":2021},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:a67931fc0e123fd37f30e0fdf652f62501c28c5dc34d8a9300261bb7fdba0c15","observation_id":"d833eacc-5cb3-480c-84da-9624a9678e28","resolution":{"observed_at":"2026-05-23T20:35:48.972191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:9495949b13224792c16ffaa120b8aeeea1ca3e9995b15cd2baab9a83b1319b46","observation_id":"9e2f48f9-a5c9-4394-aa23-73f9e3ba91ee","resolution":{"observed_at":"2026-05-23T20:33:25.591281Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Intonation and emotion: influence of pitch levels and contour type on creating emotions","venue":null,"work_id":"c2aaf000-4ab5-42d9-87d5-28498aa775de","year":2011},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:c21f366c83cfaf647b1365f9776bdd5e92841b01339a8f529f6a9f1f3f224889","observation_id":"ee69e7e0-17f7-4555-b2e8-474e91aa36a8","resolution":{"observed_at":"2026-05-23T20:35:48.922404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Analysis of emotionally salient aspects of fundamental frequency for emotion detection","venue":null,"work_id":"d87f6e00-813c-48bb-98bf-fc28eb743ad1","year":2009},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:718db364895ce5792e9fe72cf19ab435988fe60cd9d716001fbd05c1b04a7100","observation_id":"113cc4ef-a5d3-44af-8655-3f51c634c89c","resolution":{"observed_at":"2026-05-23T20:35:48.919852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pitch in emotional speech and emotional speech recognition using pitch frequency","venue":null,"work_id":"9fc82e1f-0dbc-4807-8314-4ed2b68e0b7d","year":2010},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:c0b056ed7d08ec439549f6b3b7303bc5d9f03b69e22047f0f4cdff8186f8cc31","observation_id":"c28df821-9ee5-4f54-800d-10863e76941a","resolution":{"observed_at":"2026-05-23T20:35:48.917094Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Communicating emotion: The role of prosodic features","venue":null,"work_id":"ffc3b958-4f51-4284-8f6b-134f8a42b97c","year":1985},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:12c2d682181e854b5ce4a70cf911a0c3d8b991d0c87c53810c48bb23bbbcf3c4","observation_id":"9efe4297-055b-4a9f-89ab-2dacc6fb56ca","resolution":{"observed_at":"2026-05-23T20:35:48.914058Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The global k-means clustering algorithm","venue":null,"work_id":"97b24990-3346-4516-a161-56fc3273a9e7","year":2003},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:04a3ffec6f3b3bb33e10610e46acbe3981e779a0b80a06a3253e7f8083036b6d","observation_id":"855181f3-0960-4072-91bf-c1a34991a917","resolution":{"observed_at":"2026-05-23T20:35:48.910941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generalized end-to-end loss for speaker verification","venue":null,"work_id":"bde1f965-2cce-4afa-a9d8-1677b49cbedf","year":2018},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:14358a0e1f186fd28137226c13665dbb95ae41c2b584cee1bd4f2b23366ac956","observation_id":"63f5efe2-bfd4-4afc-b7a7-6756df737bc0","resolution":{"observed_at":"2026-05-23T20:35:48.978874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wavlm: Large-scale self-supervised pre- training for full stack speech processing","venue":null,"work_id":"f4721619-0023-4e5a-aaf2-7e3288cff7e2","year":2022},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:6cb4168239af4b2b7d22045af2b0ec2bc980b8da79a640db4d2042ba387d4b3b","observation_id":"1ee941bb-36ce-4200-8d2a-43f8bc2d17a1","resolution":{"observed_at":"2026-05-23T20:35:48.975463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15185","last_updated":"2023-12-23T07:46:55Z","snapshot_observed_at":"2026-08-05T01:11:36.813583Z","submitted_at":"2023-12-23T07:46:55Z","title":"emotion2vec: Self-Supervised Pre-Training for Speech Emotion Representation","version":1},"cited_work":{"arxiv_id":"2312.15185","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.15185","snapshot_observed_at":"2026-07-03T18:28:48.480697Z","title":"emotion2vec: Self-supervised pre-training for speech emotion repre- sentation","venue":null,"work_id":"47565141-2d61-4ce8-98eb-f2c76f40fe37","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2312.15185","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:3dcac72b61211e567ebd20ced413131a5e9cf89d8967cebbfad8d062ed8c7dd6","observation_id":"8c1e784a-8609-4a47-b1bb-6dc4731fc21d","resolution":{"observed_at":"2026-05-23T20:33:25.612916Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improv- ing prosody for cross-speaker style transfer by semi-supervised style extractor and hierarchical modeling in speech synthesis","venue":null,"work_id":"2707f948-678b-4536-a98a-6099876532d4","year":2023},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:0ca120d983a59c4ea3a65fa63267f2dd864227b539024ce06eecfaf848644144","observation_id":"005b0f11-cf72-4691-8592-be398e32260c","resolution":{"observed_at":"2026-05-23T20:35:48.968859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minilm: Deep self-attention distillation for task-agnostic compression of pre- trained transformers","venue":null,"work_id":"fb17061e-c209-4e62-a56e-e0f7249265c5","year":2020},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:d8e8ba1da75ca2f72afb1c68041e31c88d2a14c78e8f7fb62d8cc2abf0786942","observation_id":"f89302da-3746-4e9b-b854-b0ed11be929d","resolution":{"observed_at":"2026-05-23T20:35:48.964677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.08317","last_updated":"2023-03-30T07:00:38Z","snapshot_observed_at":"2026-08-01T16:05:21.888603Z","submitted_at":"2022-06-16T17:24:14Z","title":"Paraformer: Fast and Accurate Parallel Transformer for Non-autoregressive End-to-End Speech Recognition","version":3},"cited_work":{"arxiv_id":"2206.08317","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2206.08317","snapshot_observed_at":"2026-07-02T01:06:23.909166Z","title":"Paraformer: Fast and accurate parallel transformer for non-autoregressive end-to-end speech recognition","venue":null,"work_id":"90141a8e-b4e7-4362-8aeb-cf0206bf13b3","year":2022},"citing_paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-23T20:31:24.494060Z"},"links":{"cited_paper":"/paper/2206.08317","citing_paper":"/paper/2409.18512"},"observation_digest":"sha256:cd44df62fbc01fd776b58a3be1789ea058c87e329243131368b95890b4eaed92","observation_id":"222b4700-42e0-42cf-a599-33585aa123cf","resolution":{"observed_at":"2026-05-23T20:33:25.602164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2409.18512","last_updated":"2026-04-03T15:54:23Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T07:46:52Z","title":"Expressive Prompting: Improving Emotion Intensity and Speaker Consistency in Zero-Shot TTS"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":11,"verified_fuzzy":22},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 3 inbound Pith citation observations for arXiv:2409.18512."}