{"as_of":"2026-08-08T22:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a3faf8ad3b94cb2a93a2968ecfcbaac3120d8d170db88f482f015863f69e9292","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":19,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:23:40.366858Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T14:38:28.620667Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2505.17589","last_updated":"2025-05-27T07:48:34Z","snapshot_observed_at":"2026-08-08T16:08:45.949013Z","submitted_at":"2025-05-23T07:55:21Z","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-16T05:27:25.425188Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2505.17589"},"observation_digest":"sha256:645cec6ee99174b644a3f0ffeb872441dcca2c71f3572250b02cb5250cde4cc7","observation_id":"0d5f7645-2d04-4d74-be77-c2ff20279371","resolution":{"observed_at":"2026-05-16T05:27:25.612037Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-07T14:23:40.366858Z","title":"Ex- presso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19103","last_updated":"2025-05-25T11:45:08Z","snapshot_observed_at":"2026-08-07T22:31:13.487819Z","submitted_at":"2025-05-25T11:45:08Z","title":"WHISTRESS: Enriching Transcriptions with Sentence Stress Detection","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:23:40.366858Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2505.19103"},"observation_digest":"sha256:5992cad70b627aa1c0411c97502dffe6bebc746e5b484c79f3d5697cef5011f2","observation_id":"742884e6-2cf3-47cd-8ac0-ae6d3c53620f","resolution":{"observed_at":"2026-08-07T14:23:40.366858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2505.22765","last_updated":"2026-04-07T15:07:17Z","snapshot_observed_at":"2026-07-06T21:32:28.358564Z","submitted_at":"2025-05-28T18:32:56Z","title":"StressTest: Can YOUR Speech LM Handle the Stress?","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-19T13:00:23.002962Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2505.22765"},"observation_digest":"sha256:f2950d27b638c3c361093c1e4c9653693039c3e07ac794af9775ad44a28266ba","observation_id":"5ab751e6-8225-4415-82ca-520f8cb83fcc","resolution":{"observed_at":"2026-05-19T13:02:18.259217Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-07T10:55:20.345371Z","title":"Ex- presso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04013","last_updated":"2025-06-04T14:42:12Z","snapshot_observed_at":"2026-08-08T13:06:14.713555Z","submitted_at":"2025-06-04T14:42:12Z","title":"Towards Better Disentanglement in Non-Autoregressive Zero-Shot Expressive Voice Conversion","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T10:55:20.345371Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2506.04013"},"observation_digest":"sha256:8e929c9fa1010c9ea44638b9bbf7d096f32e67e73f50fea976d3cdad1416e4c7","observation_id":"a5e2b151-c0e1-4e2c-beb8-922275a56e54","resolution":{"observed_at":"2026-08-07T10:55:20.345371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-06T16:33:52.438991Z","title":"Expresso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-07T19:53:51.352286Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.438991Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:da9e11b728587ea20314a4e17452c674a5247b4a399956232259809be2f5d3a8","observation_id":"79602e4a-bfe0-4eec-9975-1136cfc0f26c","resolution":{"observed_at":"2026-08-06T16:33:52.438991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-06T00:51:31.393989Z","title":"A.; Hsu, W.-N.; d'Avirro, A.; Shi, B.; Gat, I.; Fazel-Zarani, M.; Remez, T.; Copet, J.; Synnaeve, G.; Hassid, M.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04195","last_updated":"2025-08-06T08:25:26Z","snapshot_observed_at":"2026-08-06T17:34:54.346667Z","submitted_at":"2025-08-06T08:25:26Z","title":"NVSpeech: An Integrated and Scalable Pipeline for Human-Like Speech Modeling with Paralinguistic Vocalizations","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T00:51:31.393989Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2508.04195"},"observation_digest":"sha256:f5bf970dc8302c1fb1d7b1951d3f04a4a3f6c5afb08bef1e869f8db78e00bab5","observation_id":"4402f487-40f7-461f-be2d-02e40eaa173d","resolution":{"observed_at":"2026-08-06T00:51:31.393989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2509.04072","last_updated":"2026-04-21T15:32:24Z","snapshot_observed_at":"2026-07-06T22:23:48.679487Z","submitted_at":"2025-09-04T10:05:06Z","title":"Computational Narrative Understanding for Expressive Text-to-Speech","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T19:30:53.250247Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2509.04072"},"observation_digest":"sha256:b0a63d9a33be105b72333a47e6fe5bf1fa37d47fd77ed21e5448441219201a5d","observation_id":"2bbcfe48-1e89-402a-ad67-8a90aa46b3da","resolution":{"observed_at":"2026-05-18T19:31:47.053839Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2509.22220","last_updated":"2026-04-13T11:56:11Z","snapshot_observed_at":"2026-08-02T12:47:50.534468Z","submitted_at":"2025-09-26T11:32:51Z","title":"StableToken: A Noise-Robust Semantic Speech Tokenizer for Resilient SpeechLLMs","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-18T12:57:04.450462Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2509.22220"},"observation_digest":"sha256:2887292515371ee993b31325851b9b1d467212eb659a581ba9c0952b3d2231eb","observation_id":"8a8b3a10-c5d3-42aa-bd5c-3168ab4bdcfe","resolution":{"observed_at":"2026-05-18T13:01:24.344834Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-03T04:28:01.655676Z","title":"InFindings of the Association for Com- putational Linguistics: NAACL 2025, pages 1604– 1635, Albuquerque, New Mexico","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05027","last_updated":"2026-02-06T13:35:19Z","snapshot_observed_at":"2026-08-03T12:01:43.618607Z","submitted_at":"2026-02-04T20:29:16Z","title":"AudioSAE: Towards Understanding of Audio-Processing Models with Sparse AutoEncoders","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T04:28:01.655676Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2602.05027"},"observation_digest":"sha256:478a2c6a2615b31c5d12061e2cbce4de6dacb9402c64a1138f8b7c3fee29e1e3","observation_id":"0252de93-6edd-4e86-b63c-26a9a7e93e21","resolution":{"observed_at":"2026-08-03T04:28:01.655676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2603.25551","last_updated":"2026-04-06T15:20:14Z","snapshot_observed_at":"2026-07-06T22:50:41.736941Z","submitted_at":"2026-03-26T15:23:34Z","title":"Voxtral TTS","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T00:38:42.441340Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2603.25551"},"observation_digest":"sha256:8277c5acc52bd476d85cee05cd96740f98c524b1e0b82479ba8f8496fc5efd57","observation_id":"d8ea44ac-40cb-4828-8173-552871608972","resolution":{"observed_at":"2026-05-15T00:39:35.896911Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.04505","last_updated":"2026-05-06T05:18:42Z","snapshot_observed_at":"2026-08-02T05:45:46.058583Z","submitted_at":"2026-05-06T05:18:42Z","title":"JASTIN: Aligning LLMs for Zero-Shot Audio and Speech Evaluation via Natural Language Instructions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-08T16:43:33.397158Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.04505"},"observation_digest":"sha256:37e323576c5bed15c1a74b43c980f3d7971c0afc593439adb652dd49a51d0df8","observation_id":"549b17d1-640d-431b-8730-8f5d3ea72df8","resolution":{"observed_at":"2026-05-11T18:01:08.380039Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.20830","last_updated":"2026-05-20T07:21:36Z","snapshot_observed_at":"2026-08-02T05:58:18.404966Z","submitted_at":"2026-05-20T07:21:36Z","title":"Raon-OpenTTS: Open Models and Data for Robust Text-to-Speech","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T02:32:26.122526Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.20830"},"observation_digest":"sha256:7df837186ac5cb99a2c0d4a0db4912eb72da80db37d46f7fe545df3ef3cedf1b","observation_id":"0661bbb5-e44f-4264-8a08-e077c33939ac","resolution":{"observed_at":"2026-05-21T02:33:55.471353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.25504","last_updated":"2026-05-25T07:08:58Z","snapshot_observed_at":"2026-08-02T14:20:09.406007Z","submitted_at":"2026-05-25T07:08:58Z","title":"Toward Natural Emotional Text-To-Speech System with Fine-Grained Non-Verbal Expression Control","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T20:53:47.252916Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.25504"},"observation_digest":"sha256:26ec18d214cb405de6bf276e4adb2a734e8ff5bf48c16c2b89bb53f6be50d8c7","observation_id":"ff5f36a2-5395-4982-b1d5-1428c3a7f1f3","resolution":{"observed_at":"2026-06-29T20:53:57.753462Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.30748","last_updated":"2026-06-01T01:53:31Z","snapshot_observed_at":"2026-07-06T23:39:58.378823Z","submitted_at":"2026-05-29T02:25:02Z","title":"Chatterbox-Flash: Prior-Calibrated Block Diffusion for Streaming Zero-Shot TTS","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T21:31:09.399885Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.30748"},"observation_digest":"sha256:7cdf31439a5c32cef1a158c1a2bac05b709030e3704189e2c0ce76a24a346dc6","observation_id":"b73b78d3-4db9-431a-b7d5-8d14dca35cc9","resolution":{"observed_at":"2026-07-01T20:16:11.068632Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2606.01677","last_updated":"2026-06-01T04:35:28Z","snapshot_observed_at":"2026-08-01T15:52:02.822088Z","submitted_at":"2026-06-01T04:35:28Z","title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-06-28T13:17:13.510587Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2606.01677"},"observation_digest":"sha256:27adba1bb1742f57dbcb17bdd915464ca83c856e660038c48fa400458756fe6d","observation_id":"0ef35b23-59a6-4ae5-b80e-62ec4283707a","resolution":{"observed_at":"2026-07-02T00:46:24.556094Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2607.02214","last_updated":"2026-07-02T14:22:46Z","snapshot_observed_at":"2026-07-07T00:07:42.752664Z","submitted_at":"2026-07-02T14:22:46Z","title":"Unlocking Speech-Text Compositional Powers: Instruction-Following Speech Language Models without Instruction Tuning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-03T14:35:17.004680Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2607.02214"},"observation_digest":"sha256:a1af9539e0198e4e3093c7d203e5c399185675148c4fe4bfaa0d0ae45103bc7a","observation_id":"483f3ab7-4cda-43af-8e3c-9493413f6c44","resolution":{"observed_at":"2026-07-03T14:38:28.622473Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-02T02:30:34.179334Z","title":"Expresso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14310","last_updated":"2026-07-15T19:16:31Z","snapshot_observed_at":"2026-08-06T12:31:11.040445Z","submitted_at":"2026-07-15T19:16:31Z","title":"Dialogs: a studio-quality expressive conversational Russian speech corpus for dialog assistants","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T02:30:34.179334Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2607.14310"},"observation_digest":"sha256:b681a353728c2cd30f1856e3fe5adca8942409362918bddda23117edac514cc7","observation_id":"612af17a-827e-4187-b885-0b712fcafe7d","resolution":{"observed_at":"2026-08-02T02:30:34.179334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-01T07:12:10.474782Z","title":"arXiv preprint arXiv:2308.05725 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21550","last_updated":"2026-07-23T17:35:20Z","snapshot_observed_at":"2026-08-08T08:48:36.880078Z","submitted_at":"2026-07-23T17:35:20Z","title":"X$^3$-OPD: Distilling Reasoning into Large Audio-Language Models via On-Policy Alignment","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-01T07:12:10.474782Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2607.21550"},"observation_digest":"sha256:c1c9dfd24c5c1b23b8dab510227af42258506fc98a9ba2b98bd1337c23510623","observation_id":"256e3517-1a63-4ca0-b573-407c72110a94","resolution":{"observed_at":"2026-08-01T07:12:10.474782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-06T00:40:33.274250Z","title":"Ex- presso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00998","last_updated":"2026-08-02T04:54:12Z","snapshot_observed_at":"2026-08-08T21:36:52.375893Z","submitted_at":"2026-08-02T04:54:12Z","title":"Beyond One-Size-Fits-All: Personalized and Culturally Adaptive Emotional TTS via Interactive Optimization of Individual Emotion Perception Spaces","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T00:40:33.274250Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2608.00998"},"observation_digest":"sha256:597811dc90c24a5f32a85daae8b949dc78f55d7e556119d281ddca7365066768","observation_id":"fec8ed68-93ac-427b-a97e-d3dea908ede9","resolution":{"observed_at":"2026-08-06T00:40:33.274250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2308.05725/citation-record","integrity":"/paper/2308.05725/integrity","json":"/paper/2308.05725/citation-record.json","paper":"/paper/2308.05725"},"outbound":[],"paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 19 inbound Pith citation observations for arXiv:2308.05725."}