{"as_of":"2026-08-12T17:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cf96f481d205ecbab9f0f35dc2d4866498f5317e71a755751a3c3eb1cf01934d","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T07:37:44.393592Z","state":"measured"},{"denominator":47,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":47,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-03T00:29:39.365107Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-03T00:37:29.602606Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"cited_work":{"arxiv_id":"2604.16211","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16211","snapshot_observed_at":"2026-07-03T00:37:29.602606Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","venue":"cs.SD","work_id":"29275e70-04e9-49bb-9891-5d5063ccbf7f","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2604.16211","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:6a9f23b093da4e13a08e69680202a76aa1ae1fb141479bfcbf1091f128b0ab40","observation_id":"feabe402-ad24-4795-96dc-7bdb0f956477","resolution":{"observed_at":"2026-05-10T07:47:12.998128Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"cited_work":{"arxiv_id":"2604.16211","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16211","snapshot_observed_at":"2026-07-03T00:37:29.602606Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","venue":"cs.SD","work_id":"29275e70-04e9-49bb-9891-5d5063ccbf7f","year":2026},"citing_paper":{"arxiv_id":"2604.17958","last_updated":"2026-04-20T08:39:55Z","snapshot_observed_at":"2026-08-11T04:34:38.097307Z","submitted_at":"2026-04-20T08:39:55Z","title":"MINT-Bench: A Comprehensive Multilingual Benchmark for Instruction-Following Text-to-Speech","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T04:03:38.919545Z"},"links":{"cited_paper":"/paper/2604.16211","citing_paper":"/paper/2604.17958"},"observation_digest":"sha256:c3c4e2b5a5baf93191da1f8d9d2e589474aba440539c21ff69595c41e40c076f","observation_id":"930b5166-b391-40eb-a030-bdd7ab00b9fd","resolution":{"observed_at":"2026-05-11T12:11:08.463100Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"cited_work":{"arxiv_id":"2604.16211","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16211","snapshot_observed_at":"2026-07-03T00:37:29.602606Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","venue":"cs.SD","work_id":"29275e70-04e9-49bb-9891-5d5063ccbf7f","year":2026},"citing_paper":{"arxiv_id":"2607.01563","last_updated":"2026-07-02T00:43:14Z","snapshot_observed_at":"2026-08-02T20:14:44.151587Z","submitted_at":"2026-07-02T00:43:14Z","title":"Beyond Words: Towards Effective Modeling of Non-Verbal Vocalizations in ASR","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-03T00:29:39.365107Z"},"links":{"cited_paper":"/paper/2604.16211","citing_paper":"/paper/2607.01563"},"observation_digest":"sha256:6a0e022cb0db907da938ff5d95eafe300851bcda19ffd9fcbfb123a3a67b3434","observation_id":"2c026c6e-ddcd-4cd7-bd09-4bcea3cab481","resolution":{"observed_at":"2026-07-03T00:37:29.603796Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2604.16211/citation-record","integrity":"/paper/2604.16211/integrity","json":"/paper/2604.16211/citation-record.json","paper":"/paper/2604.16211"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"cited_work":{"arxiv_id":"2604.16211","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.16211","snapshot_observed_at":"2026-07-03T00:37:29.602606Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","venue":"cs.SD","work_id":"29275e70-04e9-49bb-9891-5d5063ccbf7f","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2604.16211","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:6a9f23b093da4e13a08e69680202a76aa1ae1fb141479bfcbf1091f128b0ab40","observation_id":"feabe402-ad24-4795-96dc-7bdb0f956477","resolution":{"observed_at":"2026-05-10T07:47:12.998128Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1670.0028","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"he might cough","venue":null,"work_id":"12f93b7c-3bf4-4578-a65f-32c097e0da86","year":null},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:8f4e4d6991d70c58eee16e8623f0868f0a672d46f37f5d7891601545968957b8","observation_id":"6b538caa-07c4-45ff-a6ae-0b99b6162072","resolution":{"observed_at":"2026-05-10T07:47:12.995342Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We benchmark 15 TTS systems, including 7 prompt-based and 8 tag-based systems, offering a diverse evalu- ation spectrum","venue":null,"work_id":"9a759c84-ead2-4bbd-9251-62ac65e4a1cb","year":null},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:7e496215afe61ee426e657817faacc7228caae000bc5c0e7790ad1406272ff71","observation_id":"27a5d746-ae0c-419a-96c4-2d0f50078ed0","resolution":{"observed_at":"2026-05-21T13:04:11.489824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"hallucinate","venue":null,"work_id":"091d9fcb-96fa-464e-b074-b609905c3c5c","year":null},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:7924d34ba21200b63b81855cf732f0ef7304faf2d080d627b6de2abc0707fc4e","observation_id":"1c9e028c-ee24-4c0e-8859-0b8e9adcb724","resolution":{"observed_at":"2026-05-21T13:04:11.487306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9457b9bd-8064-4129-9375-95a554f5aa8b","year":null},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:4c779c3312c0e8cf6ba74d3c03e3328cc46dff359be1309ce35df7d957532eff","observation_id":"0d710bc8-9dd7-4af6-8160-3aedcd3979f1","resolution":{"observed_at":"2026-05-21T13:04:11.447477Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"First, LLMs were used in benchmark dataset generation to draft candidate texts and speech captions, which were then reviewed, filtered, and finalized by the authors","venue":null,"work_id":"95b9be2f-ee64-4efe-8ea1-7e1fdacf0b67","year":null},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:73f7124cf5b745969772926c46a7a3f9303df6bf51c3fe42fa9834a032a19419","observation_id":"3a4da3e1-523a-4da2-a628-e8578b8e794e","resolution":{"observed_at":"2026-05-21T13:04:11.450312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Recent advances in speech language models: A sur- vey","venue":null,"work_id":"caa110f9-4fa7-48ba-8da8-95f55989ae03","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:7f45280a7f6c1e5cd159bdb8325166c6002bb87ce6e5207c0a450bbe15ddf1ac","observation_id":"04bd37d6-f0a5-4f3e-8662-e90d85a359e8","resolution":{"observed_at":"2026-05-21T13:04:11.452804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01710","last_updated":"2025-03-03T16:23:10Z","snapshot_observed_at":"2026-07-06T20:45:50.730195Z","submitted_at":"2025-03-03T16:23:10Z","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","version":1},"cited_work":{"arxiv_id":"2503.01710","doi":"10.48550/arxiv.2503.01710","metadata_source":"pith","pith_arxiv_id":"2503.01710","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","venue":"cs.SD","work_id":"31f99dad-40ae-4a19-aeff-eafa54f5b42a","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2503.01710","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:d1f5126ae824cbaa9ba9f8754589cbc1bff011e5109cd4816e4e2265af4eb038","observation_id":"7759d619-2cfb-40c0-9a1f-c3b129291c2a","resolution":{"observed_at":"2026-05-17T09:39:51.552824Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hierarchical semantic-acoustic modeling via semi-discrete residual representations for expres- sive end-to-end speech synthesis","venue":null,"work_id":"e0dc0a04-4d0d-4e39-a92a-4641c3c857b5","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:a6713c9b5ef3ca26106a896c261c291243eebf44d562eba3c35bb6535d9116d7","observation_id":"7d19110e-2473-4b44-9719-1d061dc64ff5","resolution":{"observed_at":"2026-05-21T13:04:11.449582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The geneva minimalistic acoustic parameter set (gemaps) for voice research and affective computing","venue":null,"work_id":"b363a6aa-bdb3-4821-9f0b-a31ab49cffa9","year":2015},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:96989b2440d49d3d2b2afb2f75a488a49ede9d12df57d5b29c0c031801158550","observation_id":"c380a552-f753-4b53-9953-c07da931c3c6","resolution":{"observed_at":"2026-05-21T13:04:11.456159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-10T13:40:12.749786Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2507.13155","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-07-04T07:29:38.659519Z","title":"Nonverbaltts: A public english corpus of text-aligned nonverbal vocalizations with emotion annotations for text-to-speech","venue":null,"work_id":"86df4422-0ef3-4278-b59d-371b130a40fa","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:9d53212c1786bcaac791200e076e29ef24e974f5e278f0dd752519394bdd1d11","observation_id":"8205ffe1-9fd3-4fb5-b6de-5ea8ce83faf6","resolution":{"observed_at":"2026-05-10T07:47:12.987713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.05385","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T14:25:02.431484Z","title":"A scalable pipeline for enabling non-verbal speech generation and understanding","venue":null,"work_id":"e900e5c5-d317-479f-a7ed-77199ac289b8","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:525b086de8f08296c4d1f7207e46b94251c45b67f1a84dce6414f2aabc9ab09a","observation_id":"c649467e-f95c-4a0a-a0cc-2321bdb0e766","resolution":{"observed_at":"2026-05-10T07:47:12.985035Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"SMIIP-NV: A multi-annotation non-verbal expres- sive speech corpus in mandarin for llm-based speech synthesis","venue":null,"work_id":"ae5f93d4-daba-498d-ada0-e3aa487aafca","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:b77fbebdc62879eac5a4bc2113727d7544bce679d79416659d7b7b52fc9eb756","observation_id":"af5e7410-cce0-45a0-8c47-a190754cd14a","resolution":{"observed_at":"2026-05-21T13:04:11.439499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.04195","last_updated":"2025-08-06T08:25:26Z","snapshot_observed_at":"2026-08-06T17:34:54.346667Z","submitted_at":"2025-08-06T08:25:26Z","title":"NVSpeech: An Integrated and Scalable Pipeline for Human-Like Speech Modeling with Paralinguistic Vocalizations","version":1},"cited_work":{"arxiv_id":"2508.04195","doi":null,"metadata_source":"pith","pith_arxiv_id":"2508.04195","snapshot_observed_at":"2026-07-08T14:25:02.434510Z","title":"Nvspeech: An integrated and scalable pipeline for human-like speech modeling with paralinguistic vocalizations","venue":"cs.SD","work_id":"c4a043ef-1a77-44f9-acec-11c81b7809e4","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2508.04195","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:fd43ab1e4130ed369acdf6237695c47192bc58817b2965f0a6fa26987b1bfbb1","observation_id":"610ee6e7-2250-4543-8895-9e5295d4fd57","resolution":{"observed_at":"2026-05-10T07:47:13.005685Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.18196","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Yi Ren, Chenxu Hu, Xu Tan, Tao Qin, Sheng Zhao, Zhou Zhao, and Tie-Yan Liu","venue":null,"work_id":"0389fe5e-c69e-43ed-9615-f75a9f07a02a","year":2021},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:1836d5a9659c6d9426a8fd3fe2dc3057d8e5a4166d8e127379389c6c2094d563","observation_id":"1a300b4d-37a8-4b0e-b5f9-196622d66bca","resolution":{"observed_at":"2026-05-10T07:47:13.008385Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.04508","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T00:37:29.605103Z","title":"Wesr: Scaling and evaluating word-level event-speech recognition","venue":null,"work_id":"26fbe00f-fc07-46a3-841e-c33830ba03a3","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:be926ca605e9467e5a3d80d2699ded34eb5ece0f85ac23ab94f3c452168ab6bb","observation_id":"c73c6368-2941-41f0-a830-699ceb01ff36","resolution":{"observed_at":"2026-05-10T07:47:13.003199Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.09349","last_updated":"2024-11-14T10:49:17Z","snapshot_observed_at":"2026-08-03T08:23:35.039894Z","submitted_at":"2024-11-14T10:49:17Z","title":"ParaLBench: A Large-Scale Benchmark for Computational Paralinguistics over Acoustic Foundation Models","version":1},"cited_work":{"arxiv_id":"2411.09349","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.09349","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ParaLBench: A large-scale benchmark for compu- tational paralinguistics over acoustic foundation models","venue":null,"work_id":"e14243d6-0bec-4bd4-9204-8cc1f6aeabbf","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2411.09349","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:ea61fb60fba025a9d66ba631dc49bca48ebec88de39d5089112e86a322ef1adf","observation_id":"9300312c-b63f-415f-9138-e551ad49a83d","resolution":{"observed_at":"2026-05-10T07:47:13.000667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.16381","last_updated":"2025-06-19T15:08:01Z","snapshot_observed_at":"2026-08-06T23:41:05.433184Z","submitted_at":"2025-06-19T15:08:01Z","title":"InstructTTSEval: Benchmarking Complex Natural-Language Instruction Following in Text-to-Speech Systems","version":1},"cited_work":{"arxiv_id":"2506.16381","doi":"10.48550/arxiv.2506.16381","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.16381","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InstructTTSEval: Benchmarking Complex Natural-Language Instruction Following in Text-to-Speech Systems","venue":"ArXiv.org","work_id":"a5cee304-7bf5-4e37-b892-b2abff4c7292","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2506.16381","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:ff6f6c631758f0875347b5710bbb29b0b2536bb3579812027fbd6261e2dbdb73","observation_id":"88bbdbc7-9e49-4a19-a837-94f66f121876","resolution":{"observed_at":"2026-05-10T07:47:12.979580Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.05085","last_updated":"2026-05-07T14:20:28Z","snapshot_observed_at":"2026-07-06T20:48:18.268075Z","submitted_at":"2025-03-07T02:07:00Z","title":"S2S-Arena: Evaluating Paralinguistic Instruction Following in Speech-to-Speech Models","version":2},"cited_work":{"arxiv_id":"2503.05085","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.05085","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"S2S-Arena: Evaluating Paralinguistic Instruction Following in Speech-to-Speech Models","venue":"cs.CL","work_id":"6c437235-2e5a-4562-9698-71d3701510db","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2503.05085","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:e4440edf505aa4d4284307ed8b3cef0ef4042087ff46d4f86a81e78e75edf7ba","observation_id":"daef691f-79c1-488f-912b-0c260306671c","resolution":{"observed_at":"2026-05-11T01:40:33.395224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.08723","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T19:00:05.346696Z","title":"Paras2s: Benchmarking and aligning spoken language models for paralinguistic- aware speech-to-speech interaction","venue":null,"work_id":"1710115e-84a4-4b58-a02c-b049fdbf9743","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:faea2305f561819efd6272628219616b4bbf9a909c14a6e88809357d06aa6924","observation_id":"393b341a-ea7e-4bca-a3b6-38667f2a7984","resolution":{"observed_at":"2026-05-10T07:47:12.990186Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.12135","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wavbench: Benchmarking reasoning, colloquialism, and paralinguistics for end-to-end spoken dialogue models","venue":null,"work_id":"6ad9b7ef-e7c0-4a12-941d-f4392c200549","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:34280bc484017f4458a56fd66197772c250437350db17e4fdedf0f016b76e277","observation_id":"8bd03de5-5c01-4ae1-b85d-d1ef69b6ca18","resolution":{"observed_at":"2026-05-10T07:47:12.977079Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.15352","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T00:37:29.599444Z","title":"Nv-bench: Benchmark of nonverbal vocalization synthesis for expressive text-to-speech generation","venue":null,"work_id":"14634906-3c32-4e2e-9796-4eebafc3ccc5","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:d4e5fd65dc60ef5b04ea94364a50eca25a28199abe3481322d6d4fa8c57a6847","observation_id":"1c39a316-e22e-4a4b-8b19-459207c9c1ba","resolution":{"observed_at":"2026-05-10T07:47:12.992878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ChatTTS: A generative speech model for daily dialogue","venue":null,"work_id":"60a6e9f1-7b13-4f66-a90d-3f7de098dc2a","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:1dffe98a4d0d219a4a8f88e9ded71a60216cfd843e6e9879edcf919d54bf1998","observation_id":"de8e9c7b-de27-4630-beed-ec1dcf749c1d","resolution":{"observed_at":"2026-05-21T13:04:11.437330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Higgs Audio: Text-audio foundation model from bo- son ai","venue":null,"work_id":"78f035aa-34bc-420f-9741-69bf29bbb0ba","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:0bfa584b69ab62e4592725d7a7835d9a569af8fc41a0b0b1f8b4900f95a95834","observation_id":"b2ec56f0-efba-4e04-a19a-27d88d035169","resolution":{"observed_at":"2026-05-21T13:04:11.454989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bark: Text-prompted generative audio model","venue":null,"work_id":"04eca0c3-7d1c-4176-a24b-7b96b59aa2b2","year":2023},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:8fe9997fd1d397fcaf92639050e41dea73fe97d8ad8c733f60c4b0aa6b07a8d9","observation_id":"d074fde2-2cb9-4c57-8f15-3aca76c04028","resolution":{"observed_at":"2026-05-21T13:04:11.457529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.01156","last_updated":"2024-11-09T05:58:10Z","snapshot_observed_at":"2026-08-10T00:08:15.031162Z","submitted_at":"2024-11-02T07:04:02Z","title":"Fish-Speech: Leveraging Large Language Models for Advanced Multilingual Text-to-Speech Synthesis","version":2},"cited_work":{"arxiv_id":"2411.01156","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.01156","snapshot_observed_at":"2026-07-04T07:39:38.706029Z","title":"Fish-speech: Lever- aging large language models for advanced multilingual text-to- speech synthesis","venue":null,"work_id":"10ece45c-e63a-4ab2-b106-971d79be6fba","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2411.01156","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:941e771e36b4e5fffa8362c571c2770236ab3125c83027a85f8e55516faa4764","observation_id":"e0349b6f-db83-4856-a82b-da36a5808452","resolution":{"observed_at":"2026-05-10T07:47:12.969453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Orpheus-TTS: Towards human-sounding speech","venue":null,"work_id":"11ec350b-991d-40ca-b80b-e86c5c3d3c97","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:ad63805eb2e58d6f2df9a50775914bee15584d603d59653a3db5068f840a048f","observation_id":"5e60e40e-065c-47b2-aa30-845c6eb56194","resolution":{"observed_at":"2026-05-21T13:04:11.459856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-09T04:36:59.879757Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":"2412.10117","doi":"10.48550/arxiv.2412.10117","metadata_source":"pith","pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","venue":"cs.SD","work_id":"3af84775-3b81-4078-b553-52739aae03ba","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:73280aac3b3b64922340faf44f0595d20262418246779c60a6082b4ae9ef3781","observation_id":"3e8b8a52-24ab-41f1-9ce3-a631bd20d96a","resolution":{"observed_at":"2026-05-13T06:19:09.677777Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:18:49.998415+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:18:49.998415+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Elevenlabs documentation: Models","venue":null,"work_id":"5d9f9dcf-6f0b-47fb-92de-66e421113e70","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:d2e8706ca31aced93912f4b8e166d0fabe48c3de5e0a668eee5892113982c102","observation_id":"2b61a442-c3d0-4e66-a538-8b5a990ed55b","resolution":{"observed_at":"2026-05-21T13:04:11.492174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dia: A TTS model capable of generating ultra- realistic dialogue in one pass","venue":null,"work_id":"1f28df69-db92-425e-a58b-8658a596f357","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:8a0491ba82f04b1bd392d78a729af292998ad0e8c48746ef6e4d34013ee153d9","observation_id":"62ac1692-a77a-4f85-994e-a04cb5541b5b","resolution":{"observed_at":"2026-05-21T13:04:11.484376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"SynParaSpeech: Automated synthesis of paralinguistic datasets for speech generation and understand- ing","venue":null,"work_id":"e4d1a2fe-aa50-4755-9302-09531aad54dd","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:abc573c199d8178e2b4822ac85bc864fde0933f4373375a6e81246ca185b84ee","observation_id":"afc391e5-679b-4016-80c9-e2ccbf4b9025","resolution":{"observed_at":"2026-05-21T13:04:11.489607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.02863","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T20:34:09.725065Z","title":"of of the idea that has been the same idea for a thousand years that they believe that—","venue":null,"work_id":"f1d8a605-b28a-4310-8c0c-409cbf2ca718","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:72c1ff4d0ef26cf42fef2c492efd3293f84bfa7195cea0e4662c223f2dbf3f0b","observation_id":"8d936b37-7484-4d40-83bc-5c1754f5b965","resolution":{"observed_at":"2026-05-10T07:47:12.964311Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Acoustics of breath noises in human speech: De- scriptive and three-dimensional modeling approaches","venue":null,"work_id":"cc6337a2-37d3-473a-ae21-877612ed9da1","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:9e1c177e5ef251d6f3d472cae50072dfa8c24b96d954878065f1fecab5ca3e93","observation_id":"f8f4ee65-caee-4fec-a210-27da1115ccfb","resolution":{"observed_at":"2026-05-21T13:04:11.491992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V oices without words: the spectrum of nonverbal vocalisations","venue":null,"work_id":"049bd172-21b2-464e-bd8e-f73ee94ff6b4","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:6e74cff89a86f620eb8c33a56c467c447ce7734c9e44f404236cbc94735be03b","observation_id":"38c3fd28-657b-44fc-8612-776e5fe1c62b","resolution":{"observed_at":"2026-05-21T13:04:11.494119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The acm multimedia 2022 computa- tional paralinguistics challenge: V ocalisations, stuttering, activ- ity, & mosquitoes","venue":null,"work_id":"de732e5c-2294-4ad3-a542-5443766d82b0","year":2022},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:4732a1a6d78611b6faa92bdb8199fdc25ff63f32e0cadc59ea468cf69077af00","observation_id":"d92b28e9-47f9-4735-8982-569207d58fa1","resolution":{"observed_at":"2026-05-21T13:04:11.477551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Acoustic analysis of several laughter types in conversational dialogues","venue":null,"work_id":"76eda392-da16-40a4-aa7c-83924deb582c","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:3c49d8e0a6b8ad4ce71a341013d7d848ee854c150dafc5da8bfa6774d1874c18","observation_id":"c4ab0335-7a53-4f59-ac6e-2cc86afa9dd4","resolution":{"observed_at":"2026-05-21T13:04:11.469699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An acoustic-prosodic analysis of laughter types","venue":null,"work_id":"eae14b8c-08ea-4109-b2ef-53db3417e63c","year":2024},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:f8276a1d6b163307f60d2dfdfe6f51621e9f01719557ce731d466366ea3f9208","observation_id":"32093497-db68-4a84-9949-c057dbc08bcf","resolution":{"observed_at":"2026-05-21T13:04:11.472508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DNSMOS P. 835: A non-intrusive perceptual objective speech quality metric to evalu- ate noise suppressors","venue":null,"work_id":"08c4878f-ea8c-4a6a-ac1b-09ac585acab4","year":2022},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:f543964fdd89f45348625cbe0307960eda37a59569e9c0b1bde6de2dada36b07","observation_id":"ceebf8f1-919a-40ef-86c2-c857d926823a","resolution":{"observed_at":"2026-05-21T13:04:11.470009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T02:53:19.201259Z","title":"Clap learning audio concepts from natural language supervision","venue":null,"work_id":"e37fa650-f1b0-4eaf-acd4-dc6b40ff762f","year":2023},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:b2b403121e3618e1455775c4510d3e16637f9ccea502a4b6be84a76348f2e421","observation_id":"98f242e3-30bc-42a4-90d1-512c6f41ff9c","resolution":{"observed_at":"2026-05-21T13:04:11.475025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:16:39.638521Z","title":"Robust speech recognition via large-scale weak supervision","venue":null,"work_id":"4fa75683-3cfa-46be-8cba-782f9289eb8c","year":2023},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:9cc7c9eb86bbfe2fdae13c15256d5eef221ab9c11f523528000dae0cf4b24c7d","observation_id":"44d53595-dc07-436a-a513-d4467bccf832","resolution":{"observed_at":"2026-05-21T13:04:11.479788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Paraformer: Fast and accurate parallel transformer for non-autoregressive end-to- end speech recognition","venue":null,"work_id":"480be090-87e1-46be-9bf6-d84c34a6f812","year":2022},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:b5766b11631d3c1c6b0d40b3872910fed66ab86020825ebf7b8ebe91d974f85b","observation_id":"92dd31a9-a5c4-4108-8991-cfbc5136c7b3","resolution":{"observed_at":"2026-05-21T13:04:11.478547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.14664","last_updated":"2026-04-16T09:15:45Z","snapshot_observed_at":"2026-08-11T03:26:54.318112Z","submitted_at":"2025-10-16T13:19:07Z","title":"SpeechLLM-as-Judges: Towards General and Interpretable Speech Quality Evaluation","version":2},"cited_work":{"arxiv_id":"2510.14664","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.14664","snapshot_observed_at":"2026-07-04T19:00:05.343905Z","title":"SpeechLLM-as-Judges: Towards General and Interpretable Speech Quality Evaluation","venue":"cs.SD","work_id":"9b72a259-4fd7-4368-86d8-7230a4c27985","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2510.14664","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:725a6f7a121917ee063b1b500f199735c25057c191af3fc9ffe7904231524485","observation_id":"daff7a43-b05c-4642-a9e5-39320103ca43","resolution":{"observed_at":"2026-05-10T07:47:12.961800Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05984","last_updated":"2025-06-06T11:05:48Z","snapshot_observed_at":"2026-08-11T15:34:57.421830Z","submitted_at":"2025-06-06T11:05:48Z","title":"Audio-Aware Large Language Models as Judges for Speaking Styles","version":1},"cited_work":{"arxiv_id":"2506.05984","doi":"10.48550/arxiv.2506.05984","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05984","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Audio-aware large language models as judges for speaking styles","venue":"ArXiv.org","work_id":"8c842325-baa3-4788-bf79-a6755d9f83fb","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2506.05984","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:3e86ca557570d5fac30484c31c9adb64ed1a4747547923115d47d9e251103635","observation_id":"28fd3e27-1afb-49b1-aa3a-c8274b0c770e","resolution":{"observed_at":"2026-05-10T07:47:12.959163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.15621","last_updated":"2026-01-22T03:51:43Z","snapshot_observed_at":"2026-08-04T12:22:34.670840Z","submitted_at":"2026-01-22T03:51:43Z","title":"Qwen3-TTS Technical Report","version":1},"cited_work":{"arxiv_id":"2601.15621","doi":"10.48550/arxiv.2601.15621","metadata_source":"pith","pith_arxiv_id":"2601.15621","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-TTS Technical Report","venue":"cs.SD","work_id":"6e1ae3e6-2062-4e44-a140-5c75409032cd","year":2026},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2601.15621","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:305fa3796e7c8e927a7c47dd726312a6b0c44e9b5478bfe3b7cb682ea808e076","observation_id":"ce21878a-9bd4-4981-b48d-c6e2bf057531","resolution":{"observed_at":"2026-05-16T19:24:56.189411Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":1,"verified_exact":18,"verified_fuzzy":23},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 3 inbound Pith citation observations for arXiv:2604.16211."}