{"as_of":"2026-08-09T15:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b4ef2667317c334e368cec409c32185868e5c15608a86e7615f150aa24822403","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T20:08:58.582356Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.11187/citation-record","integrity":"/paper/2508.11187/integrity","json":"/paper/2508.11187/citation-record.json","paper":"/paper/2508.11187"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:08.517389Z","title":"Retrieval and browsing of spoken content,","venue":null,"work_id":"77286195-8692-44f4-8b56-d1be12256151","year":2008},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:53.581070Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:6a3cf87a91e3b8120378af7e1f9989fb34f1b24ee531f6656a287b9dd3b115af","observation_id":"2987e7b9-c65c-420f-b679-ceba3ebe440a","resolution":{"observed_at":"2026-08-05T20:09:08.591689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:08.349778Z","title":"V oice-based information retrieval—how far are we from the text-based information retrieval?","venue":null,"work_id":"7394b706-f024-45bf-a6ae-6587694b45e6","year":2009},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:53.693568Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:5272211b1d1340d973024d7c1b3aa67d6a92c31f55cf71c50c656fd4d219ca36","observation_id":"bf2f4c16-bcd1-4ed3-8686-cffeca2a4677","resolution":{"observed_at":"2026-08-05T20:09:08.410347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:08.142807Z","title":"Spoken content retrieval: A survey of techniques and technologies,","venue":null,"work_id":"1d72da70-245a-4a44-aaae-ddce12080fdc","year":2012},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:53.819901Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:0fdfc417da64e0d9b9e31612997134adc1ec7f55c0e842b85c3e9af3b421902c","observation_id":"b7769f79-5289-4bb2-a1d8-24ac5c0951d7","resolution":{"observed_at":"2026-08-05T20:09:08.272085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:07.954264Z","title":"Spoken content retrieval—beyond cascading speech recognition with text retrieval,","venue":null,"work_id":"913adbed-3a0b-43b6-8f36-86b03de08a04","year":2015},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:53.970416Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:343055548358066f3c4058128bbf34eefd0040c5efa791ba6e2b755bff32e84f","observation_id":"0460dde0-cf7d-4deb-b65a-92845e325e4e","resolution":{"observed_at":"2026-08-05T20:09:08.061091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:07.710558Z","title":"SpeechDPR: End-to-end spoken passage retrieval for open-domain spoken question answering,","venue":null,"work_id":"1dfe42df-ca94-473c-81ce-e726f86c27e3","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:54.089386Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:a2de1d9d1b222c16e884af762d71f4f7bee2a4ae8ae4dacefc4d07f9077025c9","observation_id":"3a8108d8-5251-4262-924a-7044d05beca3","resolution":{"observed_at":"2026-08-05T20:09:07.834944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:07.478668Z","title":"Retrieval augmented end-to-end spoken dialog models,","venue":null,"work_id":"ec098a3a-a622-4de6-a488-6ab4c9be0fc0","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:54.211609Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:ee3362afaae40a281dc3e808135b418cd03e46314a0ab7fccfbef35f68c0239a","observation_id":"3447d916-4f3c-4c70-8c79-e20aedddd086","resolution":{"observed_at":"2026-08-05T20:09:07.599886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14727","last_updated":"2025-02-20T16:54:07Z","snapshot_observed_at":"2026-08-07T18:01:26.190847Z","submitted_at":"2025-02-20T16:54:07Z","title":"WavRAG: Audio-Integrated Retrieval Augmented Generation for Spoken Dialogue Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14727","snapshot_observed_at":"2026-08-05T20:08:54.327856Z","title":"WavRAG: Audio-integrated retrieval augmented generation for spoken dialogue models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:54.327856Z"},"links":{"cited_paper":"/paper/2502.14727","citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:9f191455f36de11b4724d9bd0338a03fcad56410f792b4eccbe3a628682a52ee","observation_id":"88292e66-930c-407c-b267-94e6aee151af","resolution":{"observed_at":"2026-08-05T20:08:54.327856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:07.291364Z","title":"Speech retrieval-augmented generation without automatic speech recognition,","venue":null,"work_id":"15741e81-ecdb-4642-aaa9-4da5d84966b8","year":2025},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:54.447022Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:efbc4890e7ce8d592f3327d1ce3304c71872bc74b144f3a7b528f88008a90f3a","observation_id":"bbbe176b-4fc8-4de3-acf0-0507ebdf67f6","resolution":{"observed_at":"2026-08-05T20:09:07.394761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:07.133634Z","title":"Prompting audios using acoustic properties for emotion representation,","venue":null,"work_id":"63ba4ce1-307f-4d98-9a2a-d4e25b1513ea","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:54.615420Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:530df4b9f7b030ddac44fbd20638ed437e83eb924e8fdc3f7d36666fbf0b3b40","observation_id":"048add9f-8558-4874-90a5-8277fe7b6ca7","resolution":{"observed_at":"2026-08-05T20:09:07.222536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:06.935812Z","title":"Large-scale contrastive language-audio pretraining with feature fusion and keyword-to-caption augmentation,","venue":null,"work_id":"658a3cdb-5d49-47c5-be25-4dd30800af8c","year":2023},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:54.731774Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:bae9c9d2cb4464d27d0e81b214e2f58afe63ceabd48afc18c691a69c8b6743f4","observation_id":"fd5d1968-4cc0-4b3e-b386-f081012e6276","resolution":{"observed_at":"2026-08-05T20:09:07.029219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:06.698108Z","title":"IEMOCAP: Interactive emotional dyadic motion capture database,","venue":null,"work_id":"c042bc51-fa3f-449a-8886-87669cf81a0e","year":2008},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:54.894348Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:0ca245e95dcd824d65006d5ffc0504a5ad1e58905cf9c58790a7aaf03a888504","observation_id":"9409ad26-5b5a-419b-a9d6-2d18f66845ab","resolution":{"observed_at":"2026-08-05T20:09:06.804722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:06.553153Z","title":"Emotional voice conversion: Theory, databases and ESD,","venue":null,"work_id":"68c0c31b-7fad-4d2b-8f11-f200e5fc4712","year":2022},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.011380Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:ddf1ccd2b4f7865c5d7a40355c23d2b8313c7e991c8c5b8dd2c40a4162c256c3","observation_id":"2b00b25b-503a-462f-b924-f2987c8e4824","resolution":{"observed_at":"2026-08-05T20:09:06.618778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:06.363671Z","title":"Expresso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":"48e4403e-692c-4882-8a35-6e4004a79f08","year":2023},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.123871Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:e9d49634e56f928fdcb91de179516706ed512540944b0d15dbb669c715288495","observation_id":"a631dcd0-1d34-4f75-9465-1b2af2119817","resolution":{"observed_at":"2026-08-05T20:09:06.446662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:06.204999Z","title":"PromptTTS: Controllable text-to-speech with text descriptions,","venue":null,"work_id":"c606a122-6b28-458d-ae93-53b69c1b77e9","year":2023},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.239997Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:9c428b858db8a5737d716494e52b6a32cfbfd3fa2f1fa8efbd44f43aecbeb256","observation_id":"383da48d-113c-467a-9679-9cf71a7c0c63","resolution":{"observed_at":"2026-08-05T20:09:06.256855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:05.960319Z","title":"PromptStyle: Controllable style transfer for text-to-speech with natural language descriptions,","venue":null,"work_id":"3109b77a-f29b-4822-9a89-8e0ed90f9d26","year":2023},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.336467Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:bb2bbfdb8f5187b6e2e1be4fbdb3cc83f2a0d56ea2a49a77b05d6fd327fa0cfb","observation_id":"4f6c3a39-8f78-4e9d-8f30-8295154507e9","resolution":{"observed_at":"2026-08-05T20:09:06.074349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:05.738725Z","title":"PromptTTS 2: Describing and generating voices with text prompt,","venue":null,"work_id":"35921f57-5385-4f62-aa84-7e9f2e4ce44d","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.462957Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:f5354079334901154e4f5e84813f30e58f74712b87c4e0cb4fec8f1583ae30a5","observation_id":"cefdac49-bc5f-4d28-b2f0-5f0bbdc2d1c4","resolution":{"observed_at":"2026-08-05T20:09:05.843066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:05.547529Z","title":"DreamV oice: Text-guided voice conversion,","venue":null,"work_id":"6bab2dea-eb1c-4cdc-a3b9-94257af28c6b","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.587642Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:f34d86c2a50bfcc644968b1773f41f5a46806fc1ee29c7960f6d59777e7c836d","observation_id":"4ca6dce2-e074-4916-aa4f-9695a3d41165","resolution":{"observed_at":"2026-08-05T20:09:05.624428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:05.331183Z","title":"StyleCap: Automatic speaking-style captioning from speech based on speech and language self-supervised learning models,","venue":null,"work_id":"ef0f4ddc-ee81-408b-81ce-17ac3ab88faf","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.706895Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:403c968d46d8fc7b9de2acaa5c2e35a6f5f1ea3ade5caadd0986198df00c6661","observation_id":"a5f17ddf-3b7c-43e4-be54-ad62714dc880","resolution":{"observed_at":"2026-08-05T20:09:05.438828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:05.030321Z","title":"LibriTTS-P: A corpus with speaking style and speaker identity prompts for text-to-speech and style captioning,","venue":null,"work_id":"e93389fd-f2b2-49a3-b1d4-7eadbc1a85d1","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:55.823325Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:e2f92196ba1f77a94edd0fb41008cb1a856620dde5af86ad6e18b4579ced729c","observation_id":"d516af7a-56b6-4212-a7c5-7033bc9a7978","resolution":{"observed_at":"2026-08-05T20:09:05.193490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:04.772224Z","title":"V ocabulary independent spoken term detection,","venue":null,"work_id":"07cdcc8b-e001-4519-82e2-1ca20ebc0c1a","year":2007},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.015717Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:ae7a12c447462acda9f211d2156912eee4fbb6a190b77affe69c2c65892f3d94","observation_id":"96f0f9f0-92d8-460f-97fd-6d22f5e94e2b","resolution":{"observed_at":"2026-08-05T20:09:04.909259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:04.478002Z","title":"Statistical lattice-based spoken document retrieval,","venue":null,"work_id":"9a4cf4c1-1b43-4a80-8211-9f356fd33fab","year":2010},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.143258Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:b26d083fead5500893b63289b267532b84e9f74b2cf91540d1eb58d703de43fe","observation_id":"7c8bf58c-9105-4f0e-84ba-6ff34ea07034","resolution":{"observed_at":"2026-08-05T20:09:04.619531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:04.231840Z","title":"Improved semantic retrieval of spoken content by document/query expansion with random walk over acoustic similarity graphs,","venue":null,"work_id":"e5604895-cf0a-42ca-9669-e6c8351f471a","year":2013},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.260715Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:af2a1013459f239c285bc68dcd02966e88175ac1e7efdb3a6bf9d5e8a7b7742b","observation_id":"e7af4a0a-085a-48da-b017-70bc70675d86","resolution":{"observed_at":"2026-08-05T20:09:04.334449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:03.809646Z","title":"Evaluating ASR output for information retrieval,","venue":null,"work_id":"2402f0bc-81cd-4f4b-b41e-ce40721942eb","year":2007},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.414580Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:5dc68487e9deedc05961e66b7e98d2ef63cb7a05da5141ffce3ac16044c558f2","observation_id":"35a12892-4b63-4ab0-a6fd-87b1bbd05521","resolution":{"observed_at":"2026-08-05T20:09:04.030441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:03.387643Z","title":"Investigating the global semantic impact of speech recognition error on spoken content collections,","venue":null,"work_id":"af475619-34f8-4767-b780-a9ba1fda7203","year":2009},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.573972Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:e5c9b1960d1301b13ee4d56338b2e54d2f1b8a2a18844af3241329a11fb82644","observation_id":"89d11070-717c-47cf-8bf1-3df3ad898e04","resolution":{"observed_at":"2026-08-05T20:09:03.622881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:02.969370Z","title":"Speech-centric information processing: An optimization-oriented approach,","venue":null,"work_id":"36da7624-fe3c-43e5-82b1-ec66bf23571c","year":2013},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.654325Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:7fe2abab6304f60db16c9c3b055fe270268fc59bca561b15ff4dd484a412329b","observation_id":"166dafa7-2cb9-4188-9026-d506fbb459bc","resolution":{"observed_at":"2026-08-05T20:09:03.133742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:02.624874Z","title":"Unsupervised spoken-term detection with spoken queries using segment-based dynamic time warping","venue":null,"work_id":"162e690e-af6e-4d7a-8f5e-57ef52864e06","year":2010},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.754424Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:962c6356e6ef552e36ebf560dc7f3bda2b00dc90677725c088aa74964c3ef327","observation_id":"e1876a5c-0e8c-4fef-b17a-322ba56511c9","resolution":{"observed_at":"2026-08-05T20:09:02.778218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:02.324423Z","title":"Memory efficient subsequence dtw for query-by-example spoken term detection,","venue":null,"work_id":"cde825e7-3b3b-4b33-87c2-d54b31ed007b","year":2013},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:56.873814Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:ff189681c98fede325e0b5dbc681a64c938cbb29c75419d5dc7082fc52c4ecb6","observation_id":"a1b05aaa-b43b-41c9-b51d-705d123fa730","resolution":{"observed_at":"2026-08-05T20:09:02.467412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:02.024261Z","title":"The spoken web search task at MediaEval 2012,","venue":null,"work_id":"7ada3cb0-d595-48ae-bb3e-e7584fe5a34b","year":2012},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.012996Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:09a4fa7e458829663aa147abad21a38ae473b4896071a217840b7687268be3cb","observation_id":"3cddcd3e-154d-49d2-bdde-aaa3697cb39e","resolution":{"observed_at":"2026-08-05T20:09:02.140897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:01.698056Z","title":"RECAP: Retrieval-augmented audio captioning,","venue":null,"work_id":"f3015f76-dedd-47f4-aa7d-bc985464f1df","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.134441Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:d6d758d74ae25cdeef9221540beae51b4161f7639826ab7f1aa64a89da51fae2","observation_id":"29f2ff46-011e-4f9a-8824-b0d76f1b53e0","resolution":{"observed_at":"2026-08-05T20:09:01.880751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:01.455341Z","title":"Beyond speaker identity: Text guided target speech extraction,","venue":null,"work_id":"035f2e20-ef4d-4890-a20d-9e51db89ff7c","year":2023},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.182972Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:5e7f2c55fcb120af020a76406ae5fcfdb60a6d9adbc50041be60f5f69a67dbdd","observation_id":"710f025a-c68c-43a0-a5c5-7e05f52846b8","resolution":{"observed_at":"2026-08-05T20:09:01.586330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:08:57.318770Z","title":"WavLM: Large-scale self-supervised pre- training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.318770Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:314ca4613155e98550a94b0ff5def03db895ba88251ab99a5125ab45040fd9cc","observation_id":"4ce8b136-65d5-40bd-b6db-cb39af5900cd","resolution":{"observed_at":"2026-08-05T20:08:57.318770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:01.076180Z","title":"emo- tion2vec: Self-supervised pre-training for speech emotion representation,","venue":null,"work_id":"42ea1588-1c59-43f7-ac47-e8db0ee0def1","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.429737Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:5f909738ec13a7dfe6d3dc7907a59a42bef1ba5388dc0f795b9008e397f98a31","observation_id":"02b9ac51-317e-4829-b155-f89905685ddb","resolution":{"observed_at":"2026-08-05T20:09:01.170207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:00.894805Z","title":"BERT: Pre- training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"7696f075-7e2b-41d0-9e61-ca1629afdd1f","year":2019},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.528474Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:9fbbc1e2020f74d904a0147b73f3a9f5b423dd9ddd5862ee79f64899b9e14e93","observation_id":"448eae5e-275a-4316-9d96-bf266f4d046b","resolution":{"observed_at":"2026-08-05T20:09:00.963076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.11692","last_updated":"2019-07-26T17:48:29Z","snapshot_observed_at":"2026-07-31T22:31:37.910868Z","submitted_at":"2019-07-26T17:48:29Z","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.11692","snapshot_observed_at":"2026-08-05T20:08:57.647214Z","title":"RoBERTa: A robustly optimized BERT pretraining approach,","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.647214Z"},"links":{"cited_paper":"/paper/1907.11692","citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:393356a9f6d7a69d08071fabb8f7356973c1ca85a5d566e70f8f5e86751239f7","observation_id":"79accd8c-d82c-4429-9aa6-6ce3e18e6b2b","resolution":{"observed_at":"2026-08-05T20:08:57.647214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:00.649467Z","title":"Exploring the limits of transfer learning with a unified text-to-text transformer,","venue":null,"work_id":"ba1a177e-d10e-4aa5-9c07-edd569226839","year":2020},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.795428Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:769bc2a6f51a48064a104f2107aa7a5b175bdec53e534d8285b4233f7e5aa3b0","observation_id":"d53a25fb-05a3-405c-8373-612ac569b576","resolution":{"observed_at":"2026-08-05T20:09:00.773425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:00.391445Z","title":"The flan collection: Designing data and methods for effective instruction tuning,","venue":null,"work_id":"6e4eb795-3171-4cf7-91e7-da3cd0ebc922","year":2023},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:57.923525Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:ecc809189e4246d40988a4a5e26ac12b670593c43ef687ec8e8474594253b949","observation_id":"f681b8d2-59e9-4d9b-9092-8f3d21cf3520","resolution":{"observed_at":"2026-08-05T20:09:00.530926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:09:00.114262Z","title":"Sentence-T5: Scalable sentence encoders from pre-trained text-to-text models,","venue":null,"work_id":"8ec74632-565e-4adf-9d95-ea3124e3ea31","year":2022},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.045198Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:885b8474f32cb6bbf85af4d307c0d762a8b32503671fe010b97526f829ee1b27","observation_id":"9b112b2f-79a3-4cd5-a356-4d8360b9f105","resolution":{"observed_at":"2026-08-05T20:09:00.271532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:08:59.825232Z","title":"Domain-adversarial training of neural networks,","venue":null,"work_id":"d7e774b0-377e-4100-b3ae-8b2dcb24e952","year":2016},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.052707Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:48c513b35256e90aa2cacff5a0b0a6c9da836cf3af7d0ca82e8d56b8666d9790","observation_id":"83195699-1641-47d1-af58-b3e32cbbb861","resolution":{"observed_at":"2026-08-05T20:08:59.969229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:08:59.588938Z","title":"Unsupervised domain adaptation by backpropagation,","venue":null,"work_id":"3b21ff52-e75c-4f6c-ba54-74dc4415a3c4","year":2015},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.056120Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:6f78db39dda480890c35a33e523169a219f5f80ca8e3a1766a02b239c255f170","observation_id":"24279dca-a0f0-444d-b756-51de2a8d3365","resolution":{"observed_at":"2026-08-05T20:08:59.689390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T20:08:58.063279Z","title":"GPT-4o system card,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.063279Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:4668fceda94313b9d9f043f34667e30d7fe30c53f56f9f27793a80dd69725a6a","observation_id":"b802835a-261b-4e40-bf7e-a3b5a91fb24e","resolution":{"observed_at":"2026-08-05T20:08:58.063279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:08:58.165179Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.165179Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:af5ab47659b2b905320f34d8c41dfc410114910626e61c892f1686a401d80a8e","observation_id":"96bdecf8-9edd-4b16-a788-e8cf996cd90e","resolution":{"observed_at":"2026-08-05T20:08:58.165179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:08:59.347407Z","title":"PromptTTS++: Controlling speaker identity in prompt-based text-to-speech using natural language descriptions,","venue":null,"work_id":"e840e4d1-6c11-4832-aafa-a32217e0f590","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.281102Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:ddd3b2acc6d3472606d8ef218576d7096b3089c5b6de50c560a04ea78c74e841","observation_id":"f5870a2b-bf04-496f-b0cd-2d8a5fdf736e","resolution":{"observed_at":"2026-08-05T20:08:59.461612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:08:58.981676Z","title":"PromotiCon: Prompt- based emotion controllable text-to-speech via prompt generation and matching,","venue":null,"work_id":"96a04be5-fe47-4eb2-a94e-eb33514dca7d","year":2024},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.359548Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:e6a6f99583a25175c965d94cf5ecda6361e9371f6d01b77a5d0436a6f2c91131","observation_id":"c3b5de4c-bd7f-4b68-b677-f8ce08f3a5e3","resolution":{"observed_at":"2026-08-05T20:08:59.165146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T20:08:58.752250Z","title":"Visualizing data using t-SNE,","venue":null,"work_id":"320155fc-60c2-4532-a4be-4046fe2fb9ae","year":2008},"citing_paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T20:08:58.582356Z"},"links":{"citing_paper":"/paper/2508.11187"},"observation_digest":"sha256:2b3520d638cded9f8f33e148cff73598017e88135421ee4d78442fcfd7a0a809","observation_id":"f6bc0f0e-80fd-42db-9cf4-d6fef5e76115","resolution":{"observed_at":"2026-08-05T20:08:58.841982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.11187","last_updated":"2025-08-15T03:38:21Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-06T02:04:39.995110Z","submitted_at":"2025-08-15T03:38:21Z","title":"Expressive Speech Retrieval using Natural Language Descriptions of Speaking Style"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":0,"verified_fuzzy":39},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 0 inbound Pith citation observations for arXiv:2508.11187."}