{"as_of":"2026-08-07T09:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b4f4be39e15b6c091b040bd12b5822ba7a38d93e947cb02d7c59dc059e882b06","coverage":[{"denominator":69,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":69,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:40:16.437239Z","state":"measured"},{"denominator":70,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":70,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:40:10.960719Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T20:40:17.636566Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.02176","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.02176","snapshot_observed_at":"2026-08-06T20:40:17.636566Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","venue":"cs.SD","work_id":"722c863c-9881-4c3d-b685-3930caa74464","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:10.960719Z"},"links":{"cited_paper":"/paper/2507.02176","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:08295bf1e565ff0b1a41450d8549d40279388e93ffa3a5f8fe688175be0cea4c","observation_id":"517abf04-5934-49ef-95b2-e23ab2515c83","resolution":{"observed_at":"2026-08-06T20:40:17.833571Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.02176/citation-record","integrity":"/paper/2507.02176/integrity","json":"/paper/2507.02176/citation-record.json","paper":"/paper/2507.02176"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:29.433403Z","title":"Virtual chara cters are expected to possess unique identities that remain consi stent across time","venue":null,"work_id":"cdec66a7-04a6-4952-9a5a-cc6ce1d70b80","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:10.892442Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:cf8b9064812beb13ea3282ee740e7a22d1f3c532b1dbe0d05385774ef340201b","observation_id":"fbd3ef56-0b7a-4d69-8241-d50fb04b296f","resolution":{"observed_at":"2026-08-06T20:40:29.587245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.02176","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.02176","snapshot_observed_at":"2026-08-06T20:40:17.636566Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","venue":"cs.SD","work_id":"722c863c-9881-4c3d-b685-3930caa74464","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:10.960719Z"},"links":{"cited_paper":"/paper/2507.02176","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:08295bf1e565ff0b1a41450d8549d40279388e93ffa3a5f8fe688175be0cea4c","observation_id":"517abf04-5934-49ef-95b2-e23ab2515c83","resolution":{"observed_at":"2026-08-06T20:40:17.833571Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:28.563475Z","title":"We explore which mark- ers are represented in some widely used ASV embeddings, and measure the effect of confounding factors","venue":null,"work_id":"2f84d85b-f750-41d0-8a43-2072a98252dd","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.033440Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:da00e360b05b0fa387f0483c86ffa70965302159a4fa36e12ae0ba5731ef11b1","observation_id":"f1f5725e-124c-47c6-963b-359f243ac434","resolution":{"observed_at":"2026-08-06T20:40:28.996348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:28.203914Z","title":"This reﬂects the lower bound of our metric, where we expect the smallest distances","venue":null,"work_id":"fd1cdc75-3651-409e-a83c-c34a8cfd0146","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.119053Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d83cdeaa769a632bd1cc3c987647a17486c058950e4525f422c36b0f0c0e11bc","observation_id":"ce4bbb54-3138-4eec-b835-4404f612eb6d","resolution":{"observed_at":"2026-08-06T20:40:28.311331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:28.052790Z","title":"This setting tests wh ether our metric can distinguish between speakers undistinguish - able with speech rate","venue":null,"work_id":"11888e26-bc00-4f1a-aee8-6ae2ba877ef8","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.225540Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:df5949010bc3a9d6587c4d84b9af64c1a94d88ed898aec23ceeda75123fe64ed","observation_id":"8bbfef3e-0a69-429e-9162-8340ecfa3475","resolution":{"observed_at":"2026-08-06T20:40:28.102856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.942508Z","title":"The results in the top section of Table 3 show that rhythm distances are signiﬁcantly larger between different speak ers, even those with similar speech rates","venue":null,"work_id":"c3837c28-f50f-4c6a-8e12-5ac167910874","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.368279Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d2b406003c40bf2298e82f9d1c2c9990eb57138ce7c2dcda73a41962f80303df","observation_id":"89f98804-a7b9-49a1-99e0-d0bacc9f6e2c","resolution":{"observed_at":"2026-08-06T20:40:27.992700Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.831350Z","title":"We showed that ASV embeddings mainly encode speech identity markers relating to anatomy (e.g","venue":null,"work_id":"1128f45e-a179-4667-9367-4a0e54004a47","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.464887Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:9f15d607864b55e0325e9542470332a3741bc25a8f196e347d19988a7ca4a4ad","observation_id":"0ec6aae4-9475-4471-a80b-ff699a3f4477","resolution":{"observed_at":"2026-08-06T20:40:27.880864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.675461Z","title":"An image is worth one word: Personalizing t ext-to- image generation using textual inversion,","venue":null,"work_id":"05960314-d5e2-42ab-a286-560553d518bc","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.537350Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8c9b71789e377680dc683d07f05a75e7eda0eac54b711b4803fed9ead8ddcfe6","observation_id":"5d5d9667-5627-4180-8746-56411436c248","resolution":{"observed_at":"2026-08-06T20:40:27.756865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.519522Z","title":"Dreambooth: Fine tuning text-to-image d iffusion models for subject-driven generation,","venue":null,"work_id":"cc2d1003-202c-40f9-b484-2b9365368e0d","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.595403Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:52af9da003083ca1b6432a7462695559412de22b1e1d4aea1dfe0c0b9c5be1b5","observation_id":"932e5017-ad30-4164-a120-a6d46a36daa7","resolution":{"observed_at":"2026-08-06T20:40:27.588974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16771","last_updated":"2024-12-28T17:42:44Z","snapshot_observed_at":"2026-08-07T05:22:26.276397Z","submitted_at":"2024-04-25T17:23:43Z","title":"ConsistentID: Portrait Generation with Multimodal Fine-Grained Identity Preserving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16771","snapshot_observed_at":"2026-08-06T20:40:11.679021Z","title":"ConsistentID: Portrait generation wit h multimodal ﬁne-grained identity preserving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.679021Z"},"links":{"cited_paper":"/paper/2404.16771","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d659904bf7787ab1991111e5d5b6946cb92218f919527c838727b66e91915f11","observation_id":"0c62ff45-0144-4f18-aa42-7e8b96b9b4f1","resolution":{"observed_at":"2026-08-06T20:40:11.679021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15677","last_updated":"2024-04-27T14:24:15Z","snapshot_observed_at":"2026-07-06T18:04:50.648027Z","submitted_at":"2024-04-24T06:15:31Z","title":"CharacterFactory: Sampling Consistent Characters with GANs for Diffusion Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15677","snapshot_observed_at":"2026-08-06T20:40:11.744848Z","title":"CharacterFactory: Sampling consistent charac- ters with gans for diffusion models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.744848Z"},"links":{"cited_paper":"/paper/2404.15677","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2aa1f51df2c81ec92bee10550a8cff490f42abd7adcb509167f9330a288e70fd","observation_id":"c89f5a9f-9e70-4ff9-a821-38f5722804a4","resolution":{"observed_at":"2026-08-06T20:40:11.744848Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.412634Z","title":"Generat ing video game scripts with style,","venue":null,"work_id":"cb7447de-ac9f-4aa1-bbd1-2d58e0c98fa1","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.852395Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:7bc9d59440bc34d64823d36cf54d971a55f35308c665710818b11b8b450825c6","observation_id":"5d17b08b-2cef-42f0-b20c-065a1b807fe3","resolution":{"observed_at":"2026-08-06T20:40:27.458551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.262995Z","title":"Meet your favorite character: Open-domai n chat- bot mimicking ﬁctional characters with only a few utterance s,","venue":null,"work_id":"d9de386f-80cf-4242-997f-9bbfa46fb784","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.916503Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:5692912005a8e819bc85b1ed5d45770a1180c9f9f9ff870a9c365e17a588f2d5","observation_id":"988925a3-6791-4cc9-b3ea-6d3176cae942","resolution":{"observed_at":"2026-08-06T20:40:27.351960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:27.083772Z","title":"Information conveyed by vow- els","venue":null,"work_id":"cd96ee68-9402-47cf-9983-0569aece4875","year":1957},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:11.991064Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2da48b9bc937c8968284b7cb5828687d1cbb2f5b09218c45855564a38489c1de","observation_id":"ba599d21-aa4a-44e9-9de5-a9c72b8282d5","resolution":{"observed_at":"2026-08-06T20:40:27.170164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.898394Z","title":"The perception of personal iden tity in speech: Evidence from the perception of twins’ speech,","venue":null,"work_id":"9579b171-5488-4e81-902f-f44edbb212c0","year":null},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.078389Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:abdc2369267bc597d9f548f831c290758c0874db8720f22b088ac0775ec1b992","observation_id":"d6d6a64a-a140-4335-95bd-c7a4a8d02172","resolution":{"observed_at":"2026-08-06T20:40:26.987242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.738901Z","title":"X-V ectors: Robust DNN embeddings for speaker recognition,","venue":null,"work_id":"6c851747-3ef1-438d-98c4-f8f1c3e4227c","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.158681Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:a49b51201e7a762ed04041bf11fd0c24691e693609947ecbd6cead4753f068d2","observation_id":"e7c45ab1-d99e-4c5a-812e-3435e2d3b198","resolution":{"observed_at":"2026-08-06T20:40:26.806842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.556175Z","title":"ECAPA - TDNN: emphasized channel attention, propagation and aggre ga- tion in TDNN based speaker veriﬁcation,","venue":null,"work_id":"788bb9d8-cbda-471c-bea4-b5c2f0064e4f","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.245018Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:9a866525ea16ee11fa3689a74acbb08a26d26968cd367900bead7723e4d35ca0","observation_id":"f7af280c-6e79-496f-a69c-c5d8babb5c99","resolution":{"observed_at":"2026-08-06T20:40:26.629537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.383190Z","title":"State-of-the-art speaker recogni tion with neural network embeddings in NIST SRE18 and speakers in the wild evaluations,","venue":null,"work_id":"0dfb059a-c92a-4c8c-8493-5130801f6d8f","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.319383Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:399c43b179fa7029c2cf98f5b6b8c0d93a8a0bb0f6b38981f901f26fab91c5d7","observation_id":"cc153765-41c9-49ce-b995-37e9c0f1de4d","resolution":{"observed_at":"2026-08-06T20:40:26.482722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.230507Z","title":"Generalized end-to-end loss for speaker veriﬁca- tion,","venue":null,"work_id":"57e31d58-2a5a-4067-bb59-11b7ab69da16","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.410015Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:09087a89799c2df0b075aad872c4977f537fbe1aba1ba85cfe9235e8f90f6a36","observation_id":"cb4977ac-8be1-44ea-ac18-7e54da19fefd","resolution":{"observed_at":"2026-08-06T20:40:26.302718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2407.04291","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:17.234593Z","title":"We need variations in speech synthe- sis: Sub-center modelling for speaker embeddings,","venue":null,"work_id":"d5ccb00b-fd11-4c88-89ea-2807fa7d56e1","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.504831Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:c8b2f5d793fbf955fb2428e6ca0a463d24ace4ace858cca2b332cd86eb96ca8a","observation_id":"b5ee12e4-44fe-4f82-9642-408fa023c43c","resolution":{"observed_at":"2026-08-06T20:40:17.374269Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:26.024786Z","title":"Predictions of subjective ratings and spooﬁng as- sessments of V oice Conversion Challenge 2020 submissions,","venue":null,"work_id":"c2d7b1fe-3377-48bd-9671-eb5e783f2f5d","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.603592Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:c47c6b1c49ef11ca4bbea981cf0e9fd1449e1d74f7d06a922b8445ada8391547","observation_id":"7f5eebdb-5db5-4fd2-87f8-7537d1edc768","resolution":{"observed_at":"2026-08-06T20:40:26.121033Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.823833Z","title":"Transfer learning from speaker veriﬁcat ion to multi- speaker text-to-speech synthesis,","venue":null,"work_id":"230638e9-caf4-4ccb-ba34-5859b5bfc5af","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.679830Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:93da43da1ce31cb029c3888697f838e072c447b5add4ed88bafbf345d03cdbdf","observation_id":"b0d89eba-6eef-4933-97ed-e2d568c16f86","resolution":{"observed_at":"2026-08-06T20:40:25.902006Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.646386Z","title":"Zero-shot multi-speaker text-to-sp eech with state-of-the-art neural speaker embeddings,","venue":null,"work_id":"6160ada8-a738-4b37-9ff8-7cf96f989ca5","year":2020},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.740868Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:c731091a98ad2a587631791f337b19346b6a7e49e842c919cdb73a921dbd2a20","observation_id":"d59ce44c-d9f4-412b-bedd-6b5094c18c70","resolution":{"observed_at":"2026-08-06T20:40:25.728737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.519449Z","title":"Y ourTTS: towards zero-shot multi- speaker tts and zero-shot voice conversion for everyone,","venue":null,"work_id":"6f6a41dd-5f9e-442e-b94c-3ee95ce1c238","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.837534Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:33b91c623f869522bcaa3049257ae7ba9e3b15a2df0091337bf1d3ae252c2e1b","observation_id":"38425e51-2504-4a84-874e-c042e926c1d6","resolution":{"observed_at":"2026-08-06T20:40:25.600183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.362368Z","title":"Investigating on incorporating pret rained and learnable speaker representations for multi-speaker mult i-style text-to-speech,","venue":null,"work_id":"8a3ab763-88a4-4fa1-bc66-723eb2cb7016","year":2021},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:12.936260Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:4be87c17948230d9eb9622246b9fa5d27637140f6588f1ca149356d2f2974270","observation_id":"1cb6588a-a323-4f5c-b318-afc26a02eae4","resolution":{"observed_at":"2026-08-06T20:40:25.420748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05236","last_updated":"2025-07-22T21:32:13Z","snapshot_observed_at":"2026-07-06T20:33:01.960703Z","submitted_at":"2025-02-07T06:47:11Z","title":"Koel-TTS: Enhancing LLM based Speech Generation with Preference Alignment and Classifier Free Guidance","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05236","snapshot_observed_at":"2026-08-06T20:40:13.032052Z","title":"Koel-TTS: Enhancing LLM based speec h gen- eration with preference alignment and classiﬁer free guida nce,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.032052Z"},"links":{"cited_paper":"/paper/2502.05236","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:06133a991321fd83037a9eddc04e694b9ef22d30a1ce4c7270ecacd043d1299c","observation_id":"e3bf7555-52cb-4bf9-bd88-e292621223c0","resolution":{"observed_at":"2026-08-06T20:40:13.032052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:25.126011Z","title":"The Multi-Speaker Multi-Style V oice Clo ning Chal- lenge 2021,","venue":null,"work_id":"ea14f8aa-c09e-4b7b-926e-bbbb58282780","year":2021},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.128895Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:5e1298f0c441441ec9d7d91d1984a098ba8e3204ad4362d344f9fc4b26bc9aa1","observation_id":"01a602c7-7a3a-4992-bcca-2d20ea94f10d","resolution":{"observed_at":"2026-08-06T20:40:25.237363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00529","last_updated":"2024-03-01T13:39:56Z","snapshot_observed_at":"2026-07-06T17:38:04.972649Z","submitted_at":"2024-03-01T13:39:56Z","title":"VoxGenesis: Unsupervised Discovery of Latent Speaker Manifold for Speech Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00529","snapshot_observed_at":"2026-08-06T20:40:13.209463Z","title":"V oxGenesis: Unsupervised discovery of latent speaker manifold for speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.209463Z"},"links":{"cited_paper":"/paper/2403.00529","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:758635bae23d707a4e91a786efaa10edb5d84770316eebf9503da06a8c5c4b30","observation_id":"4857260a-648a-4d2c-80a3-34fd08fb25de","resolution":{"observed_at":"2026-08-06T20:40:13.209463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.943280Z","title":"Evaluating text-to-speech synthesis from a large discrete token-based speech language model,","venue":null,"work_id":"7a845f30-ce01-4651-bca5-9be72dc41bff","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.330030Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:57ed4add7f3fd958e16085f2c67b7869f3f429f0b6478a59e02205d0c36f1075","observation_id":"fc46cf74-cdd9-420c-b7f8-d04b70d048b7","resolution":{"observed_at":"2026-08-06T20:40:25.020012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:13.420536Z","title":"A comparison of discrete and soft speech units for improved voice conversion,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.420536Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2abd6ed85c688a96d0db200f9c70c66d2c36e4ca1520baf937dae58bc0c7439d","observation_id":"f6e68c49-0f0f-4721-b3a1-898330ee7271","resolution":{"observed_at":"2026-08-06T20:40:13.420536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.742265Z","title":"V oicebox: Text-guided multilingual univ ersal speech generation at scale,","venue":null,"work_id":"e7767a03-cfbc-4548-9014-dee1d2cf9114","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.484801Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:196bc6289c5a4752e9cb428fa5b937d17b1cf9f8074ba3a01582eabf6933980e","observation_id":"9fdce733-cd03-4761-b18d-ef7f05f91428","resolution":{"observed_at":"2026-08-06T20:40:24.864555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.467120Z","title":"Neural codec language models are zero-s hot text to speech synthesizers,","venue":null,"work_id":"9be4b23c-efa6-4880-80dc-5c487ec26790","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.544795Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:750f6a316d74948a3830c6c1357d76d180edb2f8ff4c7b329b1c10aa4dd9424b","observation_id":"67b8fbcc-5f6d-447d-bf18-4f153dc8ce3d","resolution":{"observed_at":"2026-08-06T20:40:24.567138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.278278Z","title":"The Singing V oice Conversion Chall enge 2023,","venue":null,"work_id":"8eb32c08-65c6-4306-9239-3efd59cdcc73","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.610042Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:437c380b2ff9de4cdbb3912540af9deda15db6cd66e3c161f3ff58f5b3e4aec7","observation_id":"fc5d7913-3f76-4168-90e3-16b024875a87","resolution":{"observed_at":"2026-08-06T20:40:24.362172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:24.054659Z","title":"Generative Data Augmentation Challe nge: Zero- shot speech synthesis for personalized speech enhancement ,","venue":null,"work_id":"0ff0424d-8bf8-49e4-b63d-284a3df2c414","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.678518Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:d58e51493dcedfcebe530631d1adad6846290a69a54f0976b8ef6ab6b7d4e7a7","observation_id":"f3ef1777-0b8f-4ab0-9ddc-4580ee3c651c","resolution":{"observed_at":"2026-08-06T20:40:24.160692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.819597Z","title":"Acoustic properties of voice timbre t ypes and their inﬂuence on voice classiﬁcation,","venue":null,"work_id":"0523ae2c-7bd6-493b-9b55-3d6cd2a556ba","year":1977},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.758589Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:5a36db82cfd5fd808716b5ec39deeb482b6eb0cfcfcf84ccaf45da531ef54d46","observation_id":"654da8cc-1fdc-4b24-b46c-06147b004493","resolution":{"observed_at":"2026-08-06T20:40:23.920263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.558301Z","title":"Speech production patterns in producing l inguis- tic contrasts are partly determined by individual differen ces in anatomy,","venue":null,"work_id":"728836a2-0f63-4b16-b0df-1a0b47f4de71","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.823753Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f531afecaeeed00e0ba5940a57ff0d0d9035789177090142f2ce5cd2e602fd5e","observation_id":"5c79d98a-4962-4f10-802e-f487f2340f69","resolution":{"observed_at":"2026-08-06T20:40:23.680062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.377921Z","title":null,"venue":null,"work_id":"8ba5762d-51ee-4e27-90af-cdbdacd831d3","year":1980},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.871818Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:1cec96547935e4c4b00666e49f451b94fef8aab53300f8f602c21ddfeca731f1","observation_id":"bc66290e-98e4-4bd5-981a-3190fe79369b","resolution":{"observed_at":"2026-08-06T20:40:23.460801Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:23.181344Z","title":"A study of rhythm in London: Is syllable-timing a feature of Multicultural London English ?","venue":null,"work_id":"8b6d06d3-fd28-4d6b-bc35-49b741ef9009","year":2011},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.928173Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:c35854d4cd49ad9981c42d499877ef6ba5e5c5af7a101539e49403a73205a1fa","observation_id":"cae77f0c-21b8-4b0b-8ad9-32e71334c683","resolution":{"observed_at":"2026-08-06T20:40:23.289528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.942534Z","title":"The measurement of rhythm: a comparison of Sin- gapore and British English,","venue":null,"work_id":"4243b5e5-e913-4531-b666-d966cdd39a8f","year":2001},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:13.978443Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:73a81a0c9ad1048d2387cc87c824591c74caf377434c231b1920de081f302730","observation_id":"af5d4057-27d1-48f6-961b-011f8958c4ba","resolution":{"observed_at":"2026-08-06T20:40:23.067362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.734150Z","title":"Sociophonetics of phonotactic pheno mena in french,","venue":null,"work_id":"d8df83cf-b1ee-4122-ae77-f1c66e639234","year":2015},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.042470Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:4c9db5c8af58197b18f8749b46068b0e2a7fd4a57d2f976d102a1eb7e47b977b","observation_id":"10783b19-ca12-4bc5-8432-b0df7d8a26ba","resolution":{"observed_at":"2026-08-06T20:40:22.807797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.514400Z","title":"Individual di fferences in vowel production,","venue":null,"work_id":"855efbe7-0dfe-49f8-9bd2-2cc75b3508e0","year":1993},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.108230Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2378e119a669da110a8dfae568d2dd6c46726a9bcfab52b83e0a97983f2c11c4","observation_id":"4d90a904-75f6-4cf3-8437-f4ba2f576871","resolution":{"observed_at":"2026-08-06T20:40:22.620821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.332904Z","title":"Individual differences in speech pro- duction: V oice-onset-time,","venue":null,"work_id":"7b36e520-1d7a-44e6-8f59-e27d0018e0d3","year":2000},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.167152Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:572a2202e5d4a94a8a2ace83789f33d6eb68523ef74d51d4feaee7d2ba3b6e24","observation_id":"13627fac-ceff-4ca7-9de0-37f89d1c47df","resolution":{"observed_at":"2026-08-06T20:40:22.428908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:22.136492Z","title":"The social life of phonetic s and phonology,","venue":null,"work_id":"0f1635b8-8bcb-4eba-a5e7-aed7147cf7ee","year":2006},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.225970Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f6820b956aba6b227ed9e0c8d26a4a6b40cb53843628f338cedccf0f82662a95","observation_id":"6c2ed39d-60f7-4358-bb49-930065f56271","resolution":{"observed_at":"2026-08-06T20:40:22.226893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.952085Z","title":"Prosodic rhythm and Afric an American english,","venue":null,"work_id":"3cd54b60-5db7-4cec-8e2f-73cdb31ae1fc","year":2006},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.273503Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8d166f6a2c211e4db703ae97141d325782fc5b9f8cfd7feb81da034e5c417acf","observation_id":"a0a78c3e-a471-4b32-9051-7867fd9eefd9","resolution":{"observed_at":"2026-08-06T20:40:22.055134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.800973Z","title":"La liaison sans enchaˆ ınement,","venue":null,"work_id":"63cba8b3-4a44-4a5f-99de-2574733fb0eb","year":1983},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.325252Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:187c8d64f811e71043fac259189145118fce0afd64bcdc510f980ceef682b60b","observation_id":"dd9e3107-7a7a-413a-82ef-74ec4916a3e9","resolution":{"observed_at":"2026-08-06T20:40:21.865653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.629519Z","title":"Stuart-Smith, E","venue":null,"work_id":"f87a97af-165a-4722-82a5-487909637f79","year":2014},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.381549Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8afc15345bea96e82c58b67291c113571fe34427d7a78217b22c11af1482ce62","observation_id":"f7e19809-8f4a-47d6-b320-721f6b2d4a4b","resolution":{"observed_at":"2026-08-06T20:40:21.707914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.418254Z","title":"The Blizzard Challenge 2023,","venue":null,"work_id":"818144ff-f5ae-4fde-bca0-93b633740eb2","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.437994Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:24cdfdeecc63c6be74d2f594c9252c51cbfd3f5a61478e7cab166d1a6fb3ec6c","observation_id":"f2fee590-4f65-4a33-85bb-bb5f1f5ae99f","resolution":{"observed_at":"2026-08-06T20:40:21.492072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.241899Z","title":"Reﬁning the evaluation of speech synthesis: A summ ary of the blizzard challenge 2023,","venue":null,"work_id":"5bca2b5b-7153-4047-a709-ca28b5881811","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.486042Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:58e46fa75c18b9d13e44f5c0f6d8aade3f4c4a02294209ce7e5622c377889fc9","observation_id":"7c04876d-3c13-4c25-8171-85a9405c39fb","resolution":{"observed_at":"2026-08-06T20:40:21.310864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:14.536826Z","title":"Good practices for evaluation of synthesized speech,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.536826Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:0937f0396057e58bcaa56c5ffe7ff1afb050f4dca9b034b47d88e517b3bf1ec7","observation_id":"457a1f57-16e9-4294-83bd-5d1340707f4c","resolution":{"observed_at":"2026-08-06T20:40:14.536826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:21.087978Z","title":"Stuck in the MOS pit: A critical anal ysis of MOS test methodology in TTS evaluation,","venue":null,"work_id":"8be43582-4194-4fb2-97f3-89fe20ca10b2","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.599962Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:0615f670386dc3e5420f20db51dff5d09f3255aaef282a53851da051bee9a792","observation_id":"a8594ffb-9d77-4924-a681-932f2c5a1c65","resolution":{"observed_at":"2026-08-06T20:40:21.155580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.12719","last_updated":"2025-05-27T02:40:41Z","snapshot_observed_at":"2026-08-03T19:15:50.931237Z","submitted_at":"2024-11-19T18:37:45Z","title":"Rethinking MUSHRA: Addressing Modern Challenges in Text-to-Speech Evaluation","version":3},"cited_work":{"arxiv_id":"2411.12719","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.12719","snapshot_observed_at":"2026-08-06T20:40:16.820041Z","title":"Rethinking MUSHRA: Addressing Modern Challenges in Text-to-Speech Evaluation","venue":"cs.CL","work_id":"ea42185f-ce9b-4c05-92df-4297fa0901f5","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.655226Z"},"links":{"cited_paper":"/paper/2411.12719","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:e1801b43ba4ad5a9bfc96bb862c3439eafba195ad1adeabf71209a74e7794506","observation_id":"bf0a2f81-a772-4739-9fd1-56a402a46427","resolution":{"observed_at":"2026-08-06T20:40:16.930605Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.926794Z","title":"MOS vs . AB: evaluating text-to-speech systems reliably using cluster ed stan- dard errors,","venue":null,"work_id":"ff955d57-e701-42c3-8b7f-f13a76faeae6","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.709252Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f76c1ce00357e8c2d8c934e7d63522f0076ad7fc41ef33f7500fcebb6f81f004","observation_id":"26572b08-138c-4f2f-8822-ccf26c78472f","resolution":{"observed_at":"2026-08-06T20:40:21.015992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.751152Z","title":"V oxSim: A perceptual voice similarity da taset,","venue":null,"work_id":"944f555f-f3d0-4433-8d47-f594651f96a7","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.712118Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:86f664e98596f46fd6ef744530ebecbd4820cea0cedfc1bab30751b3adcb5391","observation_id":"422c058a-9164-4776-a0b1-ed510877abc7","resolution":{"observed_at":"2026-08-06T20:40:20.835953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.590042Z","title":"Salmon: A suite for acous tic language model evaluation,","venue":null,"work_id":"d1b5710d-afcf-486b-804f-1b5f9635b60a","year":2025},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.751685Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:648d8f37d96cbb6ae6ea6b9b04dec38f822a369e91ed3a177bb346e8874051a3","observation_id":"e7fd25d3-fd91-4c7c-832e-14a34f7cd461","resolution":{"observed_at":"2026-08-06T20:40:20.663293Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.389992Z","title":"Automatic evaluation of speaker simila rity,","venue":null,"work_id":"5943ffae-c306-4802-b7af-dd6608ed46b9","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:14.887907Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:7bb59a6235603dc0a40cf59477ec924b34c80764b61b15b83e4d285e3d167088","observation_id":"82dfec55-c787-4bd5-b723-5017a25bfc68","resolution":{"observed_at":"2026-08-06T20:40:20.472116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.219809Z","title":"SVSNet: An end-to-end speaker voice simi larity assessment model,","venue":null,"work_id":"edb80c0d-0be1-47f0-a563-0882fb11666b","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.048606Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:5dfb2be9315f9517e74e11dee00673237cb784d72931b3f4edc3767e5bac3fcd","observation_id":"36a86cc9-028d-4d05-b97d-d2dfadd18548","resolution":{"observed_at":"2026-08-06T20:40:20.297397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:20.058869Z","title":"The V oxCeleb Speaker Recognition Challe nge: a retrospective,","venue":null,"work_id":"10a41c4c-d83a-47f1-a280-6c6f4a88b4c3","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.178145Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:956a16b4a053b48af7fe620f9e89ef381c3e4c9fb7f9ac08bb66b184ce89fc05","observation_id":"502ce3f8-0886-4a72-9e6a-87181406f5a4","resolution":{"observed_at":"2026-08-06T20:40:20.152485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.04624","last_updated":"2021-06-08T18:22:56Z","snapshot_observed_at":"2026-08-04T07:20:14.379561Z","submitted_at":"2021-06-08T18:22:56Z","title":"SpeechBrain: A General-Purpose Speech Toolkit","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.04624","snapshot_observed_at":"2026-08-06T20:40:15.310440Z","title":"SpeechBrain: A General-Purpose Speech To olkit,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.310440Z"},"links":{"cited_paper":"/paper/2106.04624","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:2969c4254045cc18aabf66be4746d1022e24901691cd443e44ade5fc6f244eea","observation_id":"d98022c9-0b21-4b97-a133-f6480c3df776","resolution":{"observed_at":"2026-08-06T20:40:15.310440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.907921Z","title":"WavLM: large-scale self-supervised pr e-training for full stack speech processing,","venue":null,"work_id":"2ace4b71-aa6f-4f27-8d12-486c547ddb06","year":2022},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.439938Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:6b71d50a845d551b272cbe4a937d6cf917824e8c2c0445ba5eca62215457327d","observation_id":"b2034870-5e78-47c0-8f37-9038336a86b6","resolution":{"observed_at":"2026-08-06T20:40:19.995774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.758719Z","title":"The CMU Arctic speech databa ses,","venue":null,"work_id":"531837b8-b3a7-4258-a6d9-7e5e9a73a478","year":2004},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.547259Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:a5acc21d4c11801c39dd7e60d04ac6277b980e141ec4aab00240a9b1839ae19f","observation_id":"65b8ea99-700f-4336-a54f-b31272b74d15","resolution":{"observed_at":"2026-08-06T20:40:19.842760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.613645Z","title":"L2-ARCTIC: a non-native english speech c orpus,","venue":null,"work_id":"8bb117e6-a6c8-4809-913d-0644d63fac3e","year":2018},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.631101Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:94d928273d66bf2c4b009da6945b399018d20087450c8251eddddcbd23e50ac3","observation_id":"56a9ff04-0491-43ae-bee3-ac8a523caf54","resolution":{"observed_at":"2026-08-06T20:40:19.681892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.453657Z","title":"Lib- riSpeech: an ASR corpus based on public domain audio books,","venue":null,"work_id":"ff0afb23-39e6-4997-b955-6c24026a43ff","year":2015},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.701551Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:141b2cd73b076920597d787f320ef2b3bb0bfe15f023f88a28d24f795bcefd73","observation_id":"33f54206-915c-4f9a-b19d-ddd85c8d66db","resolution":{"observed_at":"2026-08-06T20:40:19.533289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.308751Z","title":"Analysis of fun- damental frequency, jitter, shimmer and vocal intensity in chil- dren with phonological disorders,","venue":null,"work_id":"f2ce5833-b451-4500-a1d2-651015b0efe0","year":2005},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.768394Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:192c17a8fa0353d1e9261f65ec7e20c435c6096964963218d474742b8d4057e3","observation_id":"e75cb6d1-b3e3-4a99-93e8-8046229d5370","resolution":{"observed_at":"2026-08-06T20:40:19.376682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:19.125037Z","title":"V ocal acoust ic analy- sis – jitter, shimmer and HNR parameters,","venue":null,"work_id":"61e79a77-39d1-4a3d-8f00-90ec65458f7b","year":2013},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.849729Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:8ba40e20d9eeafef043b9d8e55fc21acdce671172026255f873d76aa9d5c213f","observation_id":"af70a177-120a-4554-bcd0-0fe20b521ab6","resolution":{"observed_at":"2026-08-06T20:40:19.199832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.942105Z","title":"V ariation of the acoustic parameters: f0, jitter, shimmer and alpha ratio in relation with different background noise levels,","venue":null,"work_id":"4edf17d8-d9bc-44c2-a37b-1699b9d8eaec","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:15.945168Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:22e0f401ccfc7f3962d5fe5995cc1f130e529c56df0768b6438b6d658555c75d","observation_id":"d80aa272-c9b8-486a-88f8-28b67beccaef","resolution":{"observed_at":"2026-08-06T20:40:19.053593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.648723Z","title":"Rhyth m modeling for voice conversion,","venue":null,"work_id":"488e0a1a-546b-4d09-9296-c6a91e2bc285","year":2023},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.041293Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:ceca919a2242782dba5cb9a025fbd18bc503e98285fc12ebcde7ca65aea5200a","observation_id":"6bc0d3b7-7283-4f48-948d-5ea32a83db7b","resolution":{"observed_at":"2026-08-06T20:40:18.800943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.324685Z","title":"Opensmile: the munich versatile and fast open-source audio feature extractor,","venue":null,"work_id":"f21da69c-a9bb-44b5-beb4-7e1e508fbd1b","year":2010},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.162090Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:f0dbd44f3249e7f98bc66b1f10ad74fad43d253d912153811461fc5c272ac6f9","observation_id":"f993c02a-30c5-4bb9-832e-82ceaf2ffa39","resolution":{"observed_at":"2026-08-06T20:40:18.468730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05421","last_updated":"2024-07-07T15:58:11Z","snapshot_observed_at":"2026-08-04T04:15:16.477419Z","submitted_at":"2024-07-07T15:58:11Z","title":"ASRRL-TTS: Agile Speaker Representation Reinforcement Learning for Text-to-Speech Speaker Adaptation","version":1},"cited_work":{"arxiv_id":"2407.05421","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.05421","snapshot_observed_at":"2026-08-06T20:40:16.626472Z","title":"ASRRL-TTS: Agile Speaker Representation Reinforcement Learning for Text-to-Speech Speaker Adaptation","venue":"eess.AS","work_id":"3d0f52c2-ca74-4a23-b1a2-041f5b18938c","year":2024},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.321959Z"},"links":{"cited_paper":"/paper/2407.05421","citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:caa95120f2b47531d382d09dce00aa32cc2901a7c5571a485f4b24e303b670bc","observation_id":"35d8cdf7-a228-477d-8e01-9ae7453031b1","resolution":{"observed_at":"2026-08-06T20:40:16.704034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:40:18.025054Z","title":"All about audio equalization: Solu- tions and frontiers,","venue":null,"work_id":"7945530c-957f-4c2c-9098-140941ef77b1","year":2016},"citing_paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T20:40:16.437239Z"},"links":{"citing_paper":"/paper/2507.02176"},"observation_digest":"sha256:de7f5fb8c9cf20f2467b310b195ec9a9c856f6366ed733caa4a839bedd1f63ec","observation_id":"fd0d3d8b-c6c7-42cb-afba-aeecf2eaf8f8","resolution":{"observed_at":"2026-08-06T20:40:18.192615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.02176","last_updated":"2025-07-02T22:16:42Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T20:33:29.938354Z","submitted_at":"2025-07-02T22:16:42Z","title":"Analyzing and Improving Speaker Similarity Assessment for Speech Synthesis"},"reference_resolution":{"displayed":69,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":4,"verified_fuzzy":56},"total_outbound_references":69},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 69 of 69 outbound references and 1 inbound Pith citation observation for arXiv:2507.02176."}