{"as_of":"2026-08-07T19:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cd13f665bd3229d5b4a9c3e6ed52c80de90d73701e757162c54d6b846fa83e32","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:33:52.529621Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":7,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":7,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:33:51.285557Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T07:29:38.659519Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-08-06T16:33:51.285557Z","title":"Datasets for NV-TTS Datasets containing NVs can be categorized into three cate- gories","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.285557Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:8120f997f489e380ced8251d02d21e4ff1331675c2e25ef58624fde81a11c570","observation_id":"17cbd5b6-880e-436a-acf9-86d0458a3f91","resolution":{"observed_at":"2026-08-06T16:33:51.285557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2507.13155","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-07-04T07:29:38.659519Z","title":"Nonverbaltts: A public english corpus of text-aligned nonverbal vocalizations with emotion annotations for text-to-speech","venue":null,"work_id":"86df4422-0ef3-4278-b59d-371b130a40fa","year":2025},"citing_paper":{"arxiv_id":"2604.16211","last_updated":"2026-04-21T13:32:36Z","snapshot_observed_at":"2026-08-01T22:51:55.583851Z","submitted_at":"2026-04-17T16:20:55Z","title":"NVBench: A Benchmark for Speech Synthesis with Non-Verbal Vocalizations","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T07:37:44.393592Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2604.16211"},"observation_digest":"sha256:022c93364d3fd652df268bed56a9e8f9d1756259681be8d9be1e010f6d548802","observation_id":"8205ffe1-9fd3-4fb5-b6de-5ea8ce83faf6","resolution":{"observed_at":"2026-05-10T07:47:12.987713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2507.13155","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-07-04T07:29:38.659519Z","title":"Nonverbaltts: A public english corpus of text-aligned nonverbal vocalizations with emotion annotations for text-to-speech","venue":null,"work_id":"86df4422-0ef3-4278-b59d-371b130a40fa","year":2025},"citing_paper":{"arxiv_id":"2604.17435","last_updated":"2026-04-19T13:34:52Z","snapshot_observed_at":"2026-08-03T18:14:19.780984Z","submitted_at":"2026-04-19T13:34:52Z","title":"MoVE: Translating Laughter and Tears via Mixture of Vocalization Experts in Speech-to-Speech Translation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T05:36:07.090627Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2604.17435"},"observation_digest":"sha256:82ddb9be543aaf451ed9bc5e09dc04671a44db37959f55ec88e215daa41833d1","observation_id":"d14c8b33-03b2-48a1-9a05-abfb49b738f7","resolution":{"observed_at":"2026-05-10T05:41:02.481257Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2507.13155","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-07-04T07:29:38.659519Z","title":"Nonverbaltts: A public english corpus of text-aligned nonverbal vocalizations with emotion annotations for text-to-speech","venue":null,"work_id":"86df4422-0ef3-4278-b59d-371b130a40fa","year":2025},"citing_paper":{"arxiv_id":"2605.12036","last_updated":"2026-05-12T12:19:33Z","snapshot_observed_at":"2026-08-02T08:39:14.773302Z","submitted_at":"2026-05-12T12:19:33Z","title":"Towards Fine-Grained Multi-Dimensional Speech Understanding: Data Pipeline, Benchmark, and Model","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T04:03:27.608638Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2605.12036"},"observation_digest":"sha256:082493936f5e66691435af34f577ee2e29ac063fba61e57f5d5514afd6c4efbe","observation_id":"a1a764d5-44f9-4f88-bb1f-b6412dd5cb66","resolution":{"observed_at":"2026-05-13T04:07:13.494381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2507.13155","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-07-04T07:29:38.659519Z","title":"Nonverbaltts: A public english corpus of text-aligned nonverbal vocalizations with emotion annotations for text-to-speech","venue":null,"work_id":"86df4422-0ef3-4278-b59d-371b130a40fa","year":2025},"citing_paper":{"arxiv_id":"2605.25504","last_updated":"2026-05-25T07:08:58Z","snapshot_observed_at":"2026-08-02T14:20:09.406007Z","submitted_at":"2026-05-25T07:08:58Z","title":"Toward Natural Emotional Text-To-Speech System with Fine-Grained Non-Verbal Expression Control","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T20:53:47.252916Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2605.25504"},"observation_digest":"sha256:455123a3a8711f3a104a66e552e87c5d43f2f34f40cdc52496ccc07b062ceaaf","observation_id":"5f4a74b0-e084-473d-be96-81181f1ca478","resolution":{"observed_at":"2026-06-29T20:53:57.756303Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2507.13155","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-07-04T07:29:38.659519Z","title":"Nonverbaltts: A public english corpus of text-aligned nonverbal vocalizations with emotion annotations for text-to-speech","venue":null,"work_id":"86df4422-0ef3-4278-b59d-371b130a40fa","year":2025},"citing_paper":{"arxiv_id":"2606.01804","last_updated":"2026-06-03T03:45:51Z","snapshot_observed_at":"2026-08-02T05:33:25.305210Z","submitted_at":"2026-06-01T07:21:02Z","title":"SpeechEditBench: A Bilingual Multi-Attribute Benchmark for Instruction-Guided Speech Editing","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-06-28T12:54:20.815371Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2606.01804"},"observation_digest":"sha256:c08a5da2ec05277c9b5faebccc72acc13c1e784fc6f8b8c002453ee2657b4bd1","observation_id":"728e6d83-5522-4b63-a6d7-84042f735378","resolution":{"observed_at":"2026-07-02T01:06:23.893986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2507.13155","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-07-04T07:29:38.659519Z","title":"Nonverbaltts: A public english corpus of text-aligned nonverbal vocalizations with emotion annotations for text-to-speech","venue":null,"work_id":"86df4422-0ef3-4278-b59d-371b130a40fa","year":2025},"citing_paper":{"arxiv_id":"2606.21215","last_updated":"2026-06-19T08:32:07Z","snapshot_observed_at":"2026-08-06T09:41:02.079736Z","submitted_at":"2026-06-19T08:32:07Z","title":"Speaker Identity in Non-Verbal Vocalizations: Conditional Distillation and Mixture of Experts Approach","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-26T13:27:42.302206Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2606.21215"},"observation_digest":"sha256:be462ee7b8c543914b0d57c6aebe82e18e278fbc09a5713eaaa1d55f6c6cb988","observation_id":"d27f5749-c57c-497d-a35e-0d98beaa22a0","resolution":{"observed_at":"2026-07-04T07:29:38.661026Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.13155/citation-record","integrity":"/paper/2507.13155/integrity","json":"/paper/2507.13155/citation-record.json","paper":"/paper/2507.13155"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.426838Z","title":null,"venue":null,"work_id":"99c0c656-b85f-4c24-98e0-b4b569ea9192","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.139991Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:184c5fca93b0f811a308d08fdaa30ab1ec11d43b086badb7bfce3c6b6c0456fb","observation_id":"87a6fb74-2262-4a10-b581-2b52afeccdb6","resolution":{"observed_at":"2026-08-06T16:33:53.432586Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.13155","snapshot_observed_at":"2026-08-06T16:33:51.285557Z","title":"Datasets for NV-TTS Datasets containing NVs can be categorized into three cate- gories","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.285557Z"},"links":{"cited_paper":"/paper/2507.13155","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:8120f997f489e380ced8251d02d21e4ff1331675c2e25ef58624fde81a11c570","observation_id":"17cbd5b6-880e-436a-acf9-86d0458a3f91","resolution":{"observed_at":"2026-08-06T16:33:51.285557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.408852Z","title":null,"venue":null,"work_id":"081bf617-809c-45f5-b401-ea6f1d0925de","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.418343Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:d81f2b218a92aaea1b6dd0af15b9df96ee3534db26919b7a9b9494673f1b316c","observation_id":"87b6ca65-a536-49ef-8070-21f464f6749a","resolution":{"observed_at":"2026-08-06T16:33:53.414154Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.314001Z","title":"This section provides an overview of the original data sources and statistics on NVs and emotion tags in the resulting dataset","venue":null,"work_id":"ab70be4b-4c27-47ad-8c1d-43baaaf9ffce","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.907484Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:c5306be21b7061ca09f5b77d55dac64c074eedf9def86251a1b3970a7616cf6e","observation_id":"4640025e-fd66-4e60-9cd7-9b696c71913a","resolution":{"observed_at":"2026-08-06T16:33:53.319474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.373192Z","title":null,"venue":null,"work_id":"fbc4ac78-cef9-4f20-8cdc-7adbe78ea1a5","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.642452Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:0ade16f318afe600bf61c2009b395537410d8d1509113211c22364dae968299a","observation_id":"3e6d978e-6bb3-4801-a0f5-fa607411f383","resolution":{"observed_at":"2026-08-06T16:33:53.379293Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.353274Z","title":null,"venue":null,"work_id":"e635e524-4a58-411c-915e-3b946601d297","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.740570Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:fe0035f2db6133fe66e5671d5f5066d26ae2e50cc1edd673c0a25b45bfd71b45","observation_id":"0d0ecdec-2b05-429e-87fb-814d88daded4","resolution":{"observed_at":"2026-08-06T16:33:53.358492Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.334491Z","title":null,"venue":null,"work_id":"a44e14a9-d2d0-472e-bc6f-a6aee12bc44b","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.818654Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:04ce448b6bd1f5fd6b80ce086d2d364b1d4042ea683874f288cfe6e871bb7904","observation_id":"92648493-7e94-424b-a7a2-da0930b4c6fc","resolution":{"observed_at":"2026-08-06T16:33:53.339821Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.293891Z","title":null,"venue":null,"work_id":"4f4f26c7-bea7-4129-a4ae-cdc4af9ad7aa","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.986982Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:2edb75a2b20eff30349518aeeffbc07278f83f91943823bebc2304a54586159a","observation_id":"5c900b21-5a91-4c08-b54b-ec8054b77bc8","resolution":{"observed_at":"2026-08-06T16:33:53.301177Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.270805Z","title":null,"venue":null,"work_id":"53bb668d-466a-4cc7-b944-6150fa8d31b7","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.058174Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:19a63c707d732c15691edb563f2c2446f4368634cbd3c2d0ad54b4fae7894438","observation_id":"1f2ffd44-c7bd-43c3-8263-fe3ae21e1dc6","resolution":{"observed_at":"2026-08-06T16:33:53.277487Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.252369Z","title":"Psychosocial correlates of interpersonal sensitivity: A meta-analysis,","venue":null,"work_id":"06f0a304-e1d2-46ee-8a31-5564a44bfbde","year":2009},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.135573Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:90d641a97ded1e0125af4a438c175dfad3bc67384d7fe3cbe888d0d739082804","observation_id":"136223d9-6618-4c43-a622-fe38b510f94f","resolution":{"observed_at":"2026-08-06T16:33:53.258714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.232527Z","title":"Assessing the ability to recognize facial and vocal expressions of emotion: Construction and vali- dation of the emotion recognition index,","venue":null,"work_id":"8af99f07-f139-49bb-9161-a05ffc79b532","year":2011},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.217274Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:2eb593329b240f5a9bc2df54d8680748063019dda8533c1108b34a9ba9abcfce","observation_id":"be52e254-b93e-436f-badb-81711579f6c9","resolution":{"observed_at":"2026-08-06T16:33:53.239573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-06T16:33:52.294105Z","title":"Cosyvoice: A scalable multi- lingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.294105Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:08a0cc02a19287844aaf7ec72feb78c1c6ec04a5c91fc40f22327e66bb301b8d","observation_id":"f0b5e40c-2753-4cc2-837e-64baa21c24e3","resolution":{"observed_at":"2026-08-06T16:33:52.294105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03283","last_updated":"2025-04-11T07:36:53Z","snapshot_observed_at":"2026-07-06T19:10:48.519252Z","submitted_at":"2024-09-05T06:48:02Z","title":"FireRedTTS: A Foundation Text-To-Speech Framework for Industry-Level Generative Speech Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.03283","snapshot_observed_at":"2026-08-06T16:33:52.374067Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.374067Z"},"links":{"cited_paper":"/paper/2409.03283","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:9c4c6797036390e462378158174ad060976467346c9d75c8a4d9e3b4f6a8183c","observation_id":"5f720103-5c55-497f-8569-2b43f5e60094","resolution":{"observed_at":"2026-08-06T16:33:52.374067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.206889Z","title":"Laugh now cry later: Controlling time-varying emotional states of flow- matching-based zero-shot text-to-speech,","venue":null,"work_id":"f486dcf1-92b4-49ba-b6ae-178189f1c240","year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.393943Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:2dc262629010fda5527711fa780717eb0e5ed07c82ff7b2289e931a23164b5be","observation_id":"3df9789c-880d-424f-9593-7e72cccfea57","resolution":{"observed_at":"2026-08-06T16:33:53.212913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-07T06:02:40.568271Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-06T16:33:52.399963Z","title":"Cosyvoice 2: Scalable stream- ing speech synthesis with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.399963Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:cfee155b8f6a622762daad606cd32c0ca14e76c2f8feec2ae75b0caf3aa0869b","observation_id":"6495cfa1-c7a6-4151-90ec-b9d1f291bc04","resolution":{"observed_at":"2026-08-06T16:33:52.399963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.404844Z","title":"V oxceleb: Large-scale speaker verification in the wild,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.404844Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:641f43e041988bdba163a60a1800db325fb4f3361bbdab30d62b187d5442b832","observation_id":"ec980957-acff-4d6e-bef4-2e98a8a90eea","resolution":{"observed_at":"2026-08-06T16:33:52.404844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.024866Z","title":"Open automatic speech recognition leaderboard,","venue":null,"work_id":"233082fc-0fae-4175-b946-67388cfcb907","year":2023},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.458751Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:4460c8057af30aef16bd8b622dd52a148cf6616ce614617908c51034039777aa","observation_id":"3f2a9014-9bac-4237-a7ad-f46f3a6a3ce3","resolution":{"observed_at":"2026-08-06T16:33:53.030784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.177489Z","title":"The ami meeting corpus,","venue":null,"work_id":"62e243e6-5a7c-4e96-a42b-7a09708ecdd1","year":2005},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.414853Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:acabe494c3b9038a31bb5f7a9f50971d935a45031658b39f640010cd897a8c02","observation_id":"82e7f93e-685f-4f97-8359-00dc2e24e634","resolution":{"observed_at":"2026-08-06T16:33:53.183753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.157380Z","title":"Switchboard: Telephone speech corpus for research and development,","venue":null,"work_id":"34b6583f-e9c5-477a-bb26-a09659867973","year":1992},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.419630Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:6ff048684e4a3bb257705af34077bd3d3939a48f838c1e8e99a55f7cd18129cd","observation_id":"55b09350-8484-4229-be2c-97a6204740ba","resolution":{"observed_at":"2026-08-06T16:33:53.163738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.391373Z","title":"Leveraging both the alignment and timestamps provided by the event detec- tion model, we accurately placed the non-verbal vocalizations within the transcription","venue":null,"work_id":"764c5fac-f870-43ee-9ae7-281ef7e1dde6","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:51.564497Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:e1f831442a17d67c8e379399be7bdcefcbd38a6d31574330e0b0619f89769d07","observation_id":"6f692236-a839-436e-9766-43ccf61adefc","resolution":{"observed_at":"2026-08-06T16:33:53.396771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.133717Z","title":"The fisher corpus: A resource for the next generations of speech-to-text","venue":null,"work_id":"945ff3e0-c8ee-4243-bdef-0ca9a84d7688","year":2004},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.424637Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:1e7a03ec1730570c306085b1922f6009b8af7c089296c104243b88f6f78ac751","observation_id":"88a88369-b946-4c6e-82c4-e694627d2022","resolution":{"observed_at":"2026-08-06T16:33:53.143526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.099750Z","title":"Jvnv: A corpus of japanese emotional speech with verbal content and nonverbal expressions,","venue":null,"work_id":"639db132-95e2-4366-960e-7aef34813749","year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.429489Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:fb8cd39399fb94599f20e3fcf24752a2d7ddea401fe08fec30d943ce0c5c542f","observation_id":"7c55a010-8fd3-4f22-9795-ca8f18d08223","resolution":{"observed_at":"2026-08-06T16:33:53.108280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.078921Z","title":"Naturalistic emotional speech collection paradigm with online game and its psychological and acoustical assessment,","venue":null,"work_id":"a879cacf-9eaf-4fc6-86a8-be58d1509081","year":2012},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.434745Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:b31df254bee0f2dabe4977661a2d95a15595d68acacb703599b084d2591c315a","observation_id":"ca68a4a6-b1ac-4272-a745-d6c0f2d54fa6","resolution":{"observed_at":"2026-08-06T16:33:53.084336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-06T16:33:52.438991Z","title":"Expresso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.438991Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:2f6b2313655dc74bd1278bbf3fa68c0d9e1c1b72cbf197c7ac8f414ae7894dbd","observation_id":"79602e4a-bfe0-4eec-9975-1136cfc0f26c","resolution":{"observed_at":"2026-08-06T16:33:52.438991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05361","last_updated":"2024-09-07T15:08:24Z","snapshot_observed_at":"2026-08-07T13:19:22.496186Z","submitted_at":"2024-07-07T13:24:54Z","title":"Emilia: An Extensive, Multilingual, and Diverse Speech Dataset for Large-Scale Speech Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05361","snapshot_observed_at":"2026-08-06T16:33:52.444640Z","title":"Emilia: An extensive, multilingual, and diverse speech dataset for large-scale speech generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.444640Z"},"links":{"cited_paper":"/paper/2407.05361","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:3d9a5d85eac911ee37c814b9a0814013f21371e842fcbdecdd281d1e3da1ada2","observation_id":"89261381-9e78-4f79-bcef-a759ffdab20e","resolution":{"observed_at":"2026-08-06T16:33:52.444640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.061048Z","title":"Nsv-tts: Non-speech vocalization modeling and transfer in emotional text-to-speech,","venue":null,"work_id":"7c78e3e7-fdd0-4b55-99bf-2291913b3125","year":2023},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.449897Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:33611d708680d1e1b7988014bed73c687c0db53994361de35a493c0a45111d7b","observation_id":"d9476c68-7d09-408f-9f6d-7b2693086a8e","resolution":{"observed_at":"2026-08-06T16:33:53.066327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:53.044609Z","title":"NeMo: a toolkit for Conversational AI and Large Language Models","venue":null,"work_id":"008df6b4-ead1-4f71-9287-84de15cbf834","year":null},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.454080Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:ef18b7ec9e1c7431642605de426447c9798145dd59a1b254300b5d51e84a9c60","observation_id":"7ff4ddb4-70d7-475a-8690-33ffa4198115","resolution":{"observed_at":"2026-08-06T16:33:53.049474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.09546","last_updated":"2024-11-28T19:07:47Z","snapshot_observed_at":"2026-07-06T19:15:29.988880Z","submitted_at":"2024-09-14T22:00:47Z","title":"Effective Pre-Training of Audio Transformers for Sound Event Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.09546","snapshot_observed_at":"2026-08-06T16:33:52.465078Z","title":"Effective pre-training of audio transformers for sound event detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.465078Z"},"links":{"cited_paper":"/paper/2409.09546","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:6a0252218f1cb693694a7aae7df4b89b5876035502b17371b256c653f64a6922","observation_id":"4b4dfea1-52eb-4dae-a4f3-d0f49e396936","resolution":{"observed_at":"2026-08-06T16:33:52.465078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.471032Z","title":"Audio set: An ontology and human-labeled dataset for audio events,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.471032Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:b67cc57652e2d45665f5275d2b54b06a399ffd970d6bad50e1b95b5016c60616","observation_id":"f3bba382-7a63-45b7-8951-78252b3890d8","resolution":{"observed_at":"2026-08-06T16:33:52.471032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.989434Z","title":"Montreal forced aligner: Trainable text-speech align- ment using kaldi","venue":null,"work_id":"438c809e-01a4-4dd7-a325-25fbb79ec1fb","year":2017},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.475408Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:37429bbf62743ce6a30cef0eb5371d6840a45aef589cf12cdba1b88f97c381dd","observation_id":"3bb32298-9e66-4d1c-9679-1ef34d59f9cc","resolution":{"observed_at":"2026-08-06T16:33:52.994602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.972865Z","title":"emotion2vec: Self-supervised pre-training for speech emotion representation,","venue":null,"work_id":"22c5999f-3ba2-42d7-8d0f-9a5c6dc53b4b","year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.480755Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:71bce6e8ba84d99417e7134207364724e8ae526fc265181232b2aa60472f06cc","observation_id":"8d5c5918-cc5c-4af4-a37a-a92b3612e791","resolution":{"observed_at":"2026-08-06T16:33:52.978404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.956162Z","title":"Argilla - Open-source framework for data-centric NLP,","venue":null,"work_id":"47e5bd42-074f-483a-a895-6026f5e3dd89","year":2023},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.485855Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:42f343d4bb834047ed4f42cdf4efd8c829825455b197ab3e993482b6393bbde7","observation_id":"b7bce535-03e2-4be4-999d-38d6fdc98af9","resolution":{"observed_at":"2026-08-06T16:33:52.960805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.940001Z","title":"V oxceleb: A large-scale speaker identification dataset,","venue":null,"work_id":"86162d36-a258-49ad-8001-3506d51d001b","year":2017},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.490625Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:cdd9db5ac5e80dfc97486463550fe20cfd3f27e19f9af4ccacef9dd9f898d539","observation_id":"f5b54365-d866-4887-99ac-4736b7cce07b","resolution":{"observed_at":"2026-08-06T16:33:52.944756Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.923021Z","title":"V oxceleb2: Deep speaker recognition,","venue":null,"work_id":"8e8cefd7-81a1-4a62-afe5-408a38e9bc71","year":2018},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.495045Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:cf7de313c673de4591bff12af063276e2e2d5542507a54637627c86667ab7691","observation_id":"f2a65089-afe1-4259-971c-c2671b1493e9","resolution":{"observed_at":"2026-08-06T16:33:52.927795Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.501162Z","title":"Scaling rich style-prompted text-to-speech datasets,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.501162Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:3f2bd0d15e745d49168e57aeb59b276478950c5a5d149474fb0a409e2ab94811","observation_id":"1ecf28b2-5028-4614-8e14-aa4d94bf9dd3","resolution":{"observed_at":"2026-08-06T16:33:52.501162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.904974Z","title":"Optimizing speech emotion recognition with ma- chine learning based advanced audio cue analysis,","venue":null,"work_id":"d6d5f429-3612-4cd5-8958-96cf786df917","year":2024},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.506545Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:fa0c3aa64628f987c96c734e358b144276a9e1615c2b079da63449683ddc4622","observation_id":"aca2e5ff-5c90-42d7-b446-a3ac5b38c541","resolution":{"observed_at":"2026-08-06T16:33:52.910726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.884665Z","title":"Adam: A method for stochastic optimization,","venue":null,"work_id":"c0c0f9e8-e868-4156-af66-d155a67f9394","year":2014},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.510968Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:f92e159e8158b54c6b85938e93f27f0f6c550d5786bc5c1a72be4f0b32b622e0","observation_id":"f69c1adf-6449-4cc9-ad92-2804265179dd","resolution":{"observed_at":"2026-08-06T16:33:52.890637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.515663Z","title":"Wavlm: Large-scale self- supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.515663Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:d7af1fcde8fb7a2eb69c2c6b6b802c61375bf3d01e01fca1aaaddad8178eed85","observation_id":"0c4d69da-5457-4cf7-8fb0-b835cd3116a1","resolution":{"observed_at":"2026-08-06T16:33:52.515663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.520873Z","title":"Dnsmos: A non-intrusive perceptual objective speech quality metric to evaluate noise sup- pressors,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.520873Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:3a5234062bec0acc837ec7bf4f46d63d5117cb4719b2e3bfa335ade2c8924317","observation_id":"7cc61b10-9a91-4c62-8670-c84469c55788","resolution":{"observed_at":"2026-08-06T16:33:52.520873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-06T16:33:52.525067Z","title":"Robust speech recognition via large- scale weak supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.525067Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:39aefc6513f1e63f95077556cb50683b139f23cd5ee74a3b7c3c084408134f06","observation_id":"67803a5b-440f-4008-8c96-5f47244dae01","resolution":{"observed_at":"2026-08-06T16:33:52.525067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:33:52.836928Z","title":"SciPy 1.0: Fundamental Algorithms for Scientific Computing in Python,","venue":null,"work_id":"3db049ef-7495-4d1c-ad57-436104231470","year":2020},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.529621Z"},"links":{"citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:a63db383792f2594fce15d8be62106409bdd3b8a4a81c358a303349afa36698e","observation_id":"035aa6dc-40d2-4327-891e-b82e7255f0eb","resolution":{"observed_at":"2026-08-06T16:33:52.845826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T16:26:45.969196Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":19},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 7 inbound Pith citation observations for arXiv:2507.13155."}