{"as_of":"2026-08-08T22:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c9bbb3dfd0f136479220d5534aea40edc0ffa608a5c58cf90576ac5b8d0ee085","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:23:39.522256Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:23:34.275071Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T00:23:40.027554Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"cited_work":{"arxiv_id":"2506.14226","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.14226","snapshot_observed_at":"2026-08-07T00:23:40.027554Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","venue":"cs.SD","work_id":"94210cdc-856e-41cc-99e6-7f8287d5c29b","year":2025},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:34.275071Z"},"links":{"cited_paper":"/paper/2506.14226","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:d0ca39f15a49062a91fcf3b64d3a41eac35d3c4406cd8fd1a0b33529a6b975cc","observation_id":"0b9f8a7c-1599-4d43-b0e3-545013616521","resolution":{"observed_at":"2026-08-07T00:23:40.118664Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.14226/citation-record","integrity":"/paper/2506.14226/integrity","json":"/paper/2506.14226/citation-record.json","paper":"/paper/2506.14226"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"cited_work":{"arxiv_id":"2506.14226","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.14226","snapshot_observed_at":"2026-08-07T00:23:40.027554Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","venue":"cs.SD","work_id":"94210cdc-856e-41cc-99e6-7f8287d5c29b","year":2025},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:34.275071Z"},"links":{"cited_paper":"/paper/2506.14226","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:d0ca39f15a49062a91fcf3b64d3a41eac35d3c4406cd8fd1a0b33529a6b975cc","observation_id":"0b9f8a7c-1599-4d43-b0e3-545013616521","resolution":{"observed_at":"2026-08-07T00:23:40.118664Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:46.728970Z","title":"Figure 1 illustrates the entire workflow along with implementation details","venue":null,"work_id":"3d46461f-d7d0-4352-830b-a79680e56b93","year":null},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:34.359464Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:50bae2f9eff8f15697a288f9ffcf12e964d08c9f5d087f8ee4e0e0d241ecbac3","observation_id":"9973e933-9aa8-47a3-94dc-ffd25c2d0d66","resolution":{"observed_at":"2026-08-07T00:23:46.916252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:46.346307Z","title":null,"venue":null,"work_id":"59122ec4-8556-490f-a42d-ef7d97cc13a5","year":null},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:34.476186Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:2dace62dc8c9b144979b615aafc68738b7a082faaca4259977edd729b9bd14d1","observation_id":"a44aaad9-7fe4-4cab-80fd-52a8424a1a83","resolution":{"observed_at":"2026-08-07T00:23:46.580564Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:45.998032Z","title":"In this chap- ter, we will analyze the results from four aspects: TTS models, speech durations, text content, and fusion techniques","venue":null,"work_id":"b5ee822a-d982-4c02-8f03-6bc3d2892933","year":null},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:34.652055Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:53d7c29d0b5c8dc7fd8d3f3e10f2d46e814bf6e72258db507ab2bd94be2e6b8e","observation_id":"7fa08d6c-1160-481a-b3c6-1ba4cfe359b1","resolution":{"observed_at":"2026-08-07T00:23:46.156485Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:45.625297Z","title":null,"venue":null,"work_id":"806d36d8-0f2c-400e-964a-90c2335e7784","year":null},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:34.817804Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:1c5821ab24490f91056eb70da88d2d5bb0a65bc9af4b3f08d4caa25db3f19499","observation_id":"fc16b29f-4270-402a-9526-bd1d131046f7","resolution":{"observed_at":"2026-08-07T00:23:45.791905Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:45.282178Z","title":"X-vectors: Robust dnn embeddings for speaker recognition,","venue":null,"work_id":"3c525002-8b75-4c90-b66a-120032792ba2","year":2018},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:34.955023Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:cb743de76bc3a3e5a2ad8a2a99b001af5e44defce3f59fe7dd3b031ed30c9906","observation_id":"cfdc2aa2-fad8-43ee-897c-2718a8beda32","resolution":{"observed_at":"2026-08-07T00:23:45.438945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.07143","last_updated":"2020-08-10T13:50:24Z","snapshot_observed_at":"2026-08-07T15:21:18.262728Z","submitted_at":"2020-05-14T17:02:15Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.07143","snapshot_observed_at":"2026-08-07T00:23:35.159496Z","title":"Ecapa- tdnn: Emphasized channel attention, propagation and ag- gregation in tdnn based speaker verification,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:35.159496Z"},"links":{"cited_paper":"/paper/2005.07143","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:3e0b54d89b15569a70bb8af0cfba0ce7795d96a67d343d0580dbf9a1e897207a","observation_id":"297f5257-1324-411c-8c10-a9896aae4609","resolution":{"observed_at":"2026-08-07T00:23:35.159496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.12592","last_updated":"2019-10-16T11:27:27Z","snapshot_observed_at":"2026-07-06T08:32:43.131328Z","submitted_at":"2019-10-16T11:27:27Z","title":"BUT System Description to VoxCeleb Speaker Recognition Challenge 2019","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.12592","snapshot_observed_at":"2026-08-07T00:23:35.266115Z","title":"But system description to voxceleb speaker recognition chal- lenge 2019,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:35.266115Z"},"links":{"cited_paper":"/paper/1910.12592","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:41d7296615aac2c4dc7c8f3d03a4f135cb053783890ff7e321bfd593a8196edb","observation_id":"f97ac33d-b4d3-4252-9882-0c1fa44a9ae7","resolution":{"observed_at":"2026-08-07T00:23:35.266115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:44.979288Z","title":"Mfa-conformer: Multi-scale feature aggregation con- former for automatic speaker verification,","venue":null,"work_id":"877eb710-5408-4abb-b8de-d3a23b1ac5aa","year":2022},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:35.483658Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:bfc34495cedae16ed46cb4ff5384d9121eb07e4d99bcbb82063138f330c36839","observation_id":"10c0fe75-1074-40f5-8838-8a34627ac098","resolution":{"observed_at":"2026-08-07T00:23:45.099236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:44.513157Z","title":"Memory storable network based feature aggregation for speaker representation learning,","venue":null,"work_id":"d5b9c4e1-d3b5-43be-be2d-c23c1350722a","year":2023},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:35.675854Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:64ade77e5a6c35fa4cee719a53ecb4896c39a12f897babebd96a929b9cb05176","observation_id":"24f4ab30-36b1-4060-8fde-d6ed1896065f","resolution":{"observed_at":"2026-08-07T00:23:44.808818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:44.204637Z","title":"Whisper-pmfa: Partial multi-scale feature aggregation for speaker verification using whisper models,","venue":null,"work_id":"d104246f-d9de-4d21-91de-c639a59823fa","year":2024},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:35.834899Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:67b6b7100e56b06553775ec441aedcebbf2c4075d0afb96db34484f2fea4f516","observation_id":"0ec4f438-5efa-4479-9acc-4b2a69076981","resolution":{"observed_at":"2026-08-07T00:23:44.356554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:43.856574Z","title":"Cam++: A fast and efficient network for speaker verification using context- aware masking,","venue":null,"work_id":"aee40589-7704-4fc3-8879-7bd4ed6c701c","year":2023},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:35.961690Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:7448672e6bd2cb4d35070c93f5273b604fb7353cda4c4a7fdddcbd7e981aeff5","observation_id":"8bdc73cb-126e-45c9-b724-6b5d1ba3a97c","resolution":{"observed_at":"2026-08-07T00:23:44.041649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02167","last_updated":"2024-06-04T09:56:39Z","snapshot_observed_at":"2026-07-06T18:25:05.750197Z","submitted_at":"2024-06-04T09:56:39Z","title":"ERes2NetV2: Boosting Short-Duration Speaker Verification Performance with Computational Efficiency","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02167","snapshot_observed_at":"2026-08-07T00:23:36.116032Z","title":"Eres2netv2: Boosting short-duration speaker verifica- tion performance with computational efficiency,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:36.116032Z"},"links":{"cited_paper":"/paper/2406.02167","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:d4ffeca8a4bb315136a37062e7c0fe55159bd78082e0f44eeaf747a7e4e5e855","observation_id":"151a4959-cb8f-428f-bf76-24f60cc67119","resolution":{"observed_at":"2026-08-07T00:23:36.116032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:43.470820Z","title":"Overview of speaker modeling and its applications: From the lens of deep speaker representation learning,","venue":null,"work_id":"744b7799-abea-40b1-aed1-15f759326b5b","year":2024},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:36.293870Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:b6ba34e9212f689e35ac5443e405ecafd795ef47080007954481e554834acdce","observation_id":"d30666ed-781b-4b9e-a053-dd84df28f6e6","resolution":{"observed_at":"2026-08-07T00:23:43.653995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.10420","last_updated":"2019-07-22T23:43:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-07-22T23:43:20Z","title":"A Deep Neural Network for Short-Segment Speaker Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.10420","snapshot_observed_at":"2026-08-07T00:23:36.448875Z","title":"A deep neural network for short- segment speaker recognition,","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:36.448875Z"},"links":{"cited_paper":"/paper/1907.10420","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:edecfa13acfe647892c04caa6e2a7f5b604139041dde4b821d77e998bf4872ae","observation_id":"d3bc998b-e31d-40b5-aa07-a8fa0904985b","resolution":{"observed_at":"2026-08-07T00:23:36.448875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:43.079359Z","title":"Deep speaker embedding learning with multi-level pooling for text-independent speaker verification,","venue":null,"work_id":"ce165d1b-3f7d-48f4-b52a-8a321bb97a63","year":2019},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:36.597158Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:a6349c1872bf17e9b14a40a1e7a82c4eaedf24e7f69a36c232ec7fad5d4e3959","observation_id":"ee3d4389-ef59-4375-bc69-a325f536e4a1","resolution":{"observed_at":"2026-08-07T00:23:43.255824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:42.693369Z","title":"Improving multi-scale aggregation using feature pyramid module for robust speaker verification of variable-duration utterances,","venue":null,"work_id":"f51c687e-9e26-487f-8ddf-7a324977ab27","year":2020},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:36.708396Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:7864090e23c0886b062748828a29cecca19d445adfae92405e8f5cabc8e0a86c","observation_id":"1c5f8383-47a0-4eb4-a978-36769352e694","resolution":{"observed_at":"2026-08-07T00:23:42.940747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:42.354149Z","title":"Improving aggregation and loss function for better embedding learning in end-to-end speaker verification system","venue":null,"work_id":"f19e9e59-c4c6-4473-bba5-ed133adcd04a","year":2019},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:36.907874Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:64b65f36bcc16c566ddb23e1e29f54a5c0ecf2406f650ecf3b53ae7584e03250","observation_id":"def8d12d-299e-4f01-94fb-17fcdc48ecaa","resolution":{"observed_at":"2026-08-07T00:23:42.523611Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:42.048504Z","title":"Meta- learning for short utterance speaker recognition with imbalance length pairs,","venue":null,"work_id":"9c0db883-fa6e-4f18-80c4-6ddd8823236b","year":2020},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:37.123479Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:2cd510fbfccc8b629a4d8287e2e25842836a36a5f61df30a1f6a1873a40cdc3c","observation_id":"de3e5762-cc57-4d0a-a919-a4f3bf82ac27","resolution":{"observed_at":"2026-08-07T00:23:42.158144Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:41.844106Z","title":"Text-independent speaker verification with adversarial learning on short utterances,","venue":null,"work_id":"6f299cdb-5951-4717-93df-0ea61e85d9ed","year":2020},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:37.380968Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:502dcda3b4a2d2e6b89958797299d1e1177002c8cce2fb098aba479857142c6c","observation_id":"cb082490-694a-442f-afce-26d928525058","resolution":{"observed_at":"2026-08-07T00:23:41.927984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:37.552964Z","title":"Short utterance compensation in speaker verification via cosine-based teacher- student learning of speaker embeddings,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:37.552964Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:794fba4ae8e69aed4812178748b4ca4571aca89b17c306d4160669f403364e68","observation_id":"ba0ee5bc-5b31-4819-90cf-2da52a2f0e78","resolution":{"observed_at":"2026-08-07T00:23:37.552964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:41.606826Z","title":"Open-set short utterance forensic speaker verification using teacher-student network with explicit inductive bias,","venue":null,"work_id":"31e5262f-8fd7-439f-841c-20212eec86f0","year":2020},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:37.669130Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:59dc6c9288df556b344cd95bc37046ddcfc128e1b857c44744a3a39e7b67cdf6","observation_id":"e984d1ca-eb9e-47e3-8dc6-c5de88d29acc","resolution":{"observed_at":"2026-08-07T00:23:41.719782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:41.400353Z","title":"Cnn-based joint mapping of short and long utterance i-vectors for speaker verification us- ing short utterances","venue":null,"work_id":"a4c0e2bb-9384-496f-90f9-135514d9dd56","year":2017},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:37.792093Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:e68ce5415e4b3602c2251e9908c064168821faf91f0102f8f77f3015e5a81265","observation_id":"73aa36b4-de62-4249-9755-cc41a372d421","resolution":{"observed_at":"2026-08-07T00:23:41.469790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:41.199558Z","title":"I-vector transformation us- ing conditional generative adversarial networks for short utterance speaker verification,","venue":null,"work_id":"df58ced7-a603-4257-9dd8-8fa573523d59","year":2018},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:37.963583Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:c90a3c3cdf3fc251445492a844f2dbada3a9672d6c334c484c43532df4deed4d","observation_id":"13daafee-65ca-4e1c-b3a5-a1dbea829d88","resolution":{"observed_at":"2026-08-07T00:23:41.261227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:41.033660Z","title":"Data augmentation using deep generative models for embedding based speaker recog- nition,","venue":null,"work_id":"0fd909b1-42c3-408d-9caa-478f15492d2c","year":2020},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:38.178009Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:58862c1ad27ff740cba6fcb13ec4378b77d27c7f11f0f325a08fc6e09cc2bbb9","observation_id":"2d22789f-ddf8-4565-884f-55962c6a5249","resolution":{"observed_at":"2026-08-07T00:23:41.118213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2011.10710","last_updated":"2020-11-21T02:55:47Z","snapshot_observed_at":"2026-08-08T22:49:41.017156Z","submitted_at":"2020-11-21T02:55:47Z","title":"Exploring Voice Conversion based Data Augmentation in Text-Dependent Speaker Verification","version":1},"cited_work":{"arxiv_id":"2011.10710","doi":null,"metadata_source":"pith","pith_arxiv_id":"2011.10710","snapshot_observed_at":"2026-08-07T00:23:39.770274Z","title":"Exploring Voice Conversion based Data Augmentation in Text-Dependent Speaker Verification","venue":"cs.SD","work_id":"f82c7c28-90ec-4d45-9793-05fe09aa6718","year":2020},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:38.319326Z"},"links":{"cited_paper":"/paper/2011.10710","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:ba800fd2b46aa9d83474c5369b45528c762c546c7e67d39488f89f5e6acfaf2f","observation_id":"eb19a3e5-de23-4c3e-97b2-2d27d19b2232","resolution":{"observed_at":"2026-08-07T00:23:39.839245Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:40.823814Z","title":"Synaug: Synthesis- based data augmentation for text-dependent speaker verification,","venue":null,"work_id":"7727e1b8-66aa-467c-b66b-684b910c6201","year":2021},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:38.481487Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:a4c791d710aa128847fb2907c53c5d7b4eddf9f353d0bb6d816a0529863e4539","observation_id":"27676973-d98d-4a4b-87fb-31005c1d8553","resolution":{"observed_at":"2026-08-07T00:23:40.915478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-07T00:23:38.647520Z","title":"Neural codec language mod- els are zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:38.647520Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:755d922a141b0bd55b53929e307dbfc7bce7e7967ccebf3909425c5f925ac02d","observation_id":"8ea03eae-a2b2-4f40-8d73-c1b3035e4ac6","resolution":{"observed_at":"2026-08-07T00:23:38.647520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T00:23:38.830041Z","title":"Cosyvoice: A scalable multi- lingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:38.830041Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:ae04600c302ef39048042a54b2a217b55cb86a1656e2f8da3c220eccb92ba107","observation_id":"361f4ba3-eb41-495f-acab-c941748f4b61","resolution":{"observed_at":"2026-08-07T00:23:38.830041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-07T00:23:38.994437Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:38.994437Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:a48c2eca6b668f92ebca09b28640bd9b1d60fe0305e91ed6a66d5c757301fc81","observation_id":"51d3e78e-d0b8-4bbf-804e-24cab88dc804","resolution":{"observed_at":"2026-08-07T00:23:38.994437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:39.162765Z","title":"V oxceleb: Large-scale speaker verification in the wild,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:39.162765Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:0b8692724935060a1fcf2b15030d7850fb6143b09ce02757d7e5adceb2efdbe1","observation_id":"37eb4ea3-c883-4700-810d-680a1b0903a4","resolution":{"observed_at":"2026-08-07T00:23:39.162765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:40.455288Z","title":"Wespeaker: A research and production oriented speaker embedding learning toolkit,","venue":null,"work_id":"a788d2cf-7bce-4805-b169-639149a4a093","year":2023},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:39.300553Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:85af74fcc68862cb024d520019bc48452516a60c0de9a49c7cfa828b8f60d70a","observation_id":"9061c3ca-6c0f-4f46-9b81-48cbe11d2ac7","resolution":{"observed_at":"2026-08-07T00:23:40.645490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.00750","last_updated":"2024-10-20T14:25:49Z","snapshot_observed_at":"2026-08-01T10:20:49.367508Z","submitted_at":"2024-09-01T15:26:30Z","title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.00750","snapshot_observed_at":"2026-08-07T00:23:39.416451Z","title":"Maskgct: Zero-shot text-to- speech with masked generative codec transformer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:39.416451Z"},"links":{"cited_paper":"/paper/2409.00750","citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:24e9dea3a2a0f22396261548c6aaaf84f4ceb44e633fbd9c2f093a84f2a49199","observation_id":"22e0c373-417c-4506-8d24-986b3f37aa1c","resolution":{"observed_at":"2026-08-07T00:23:39.416451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:23:40.215961Z","title":"Naturalspeech 3: Zero-shot speech syn- thesis with factorized codec and diffusion models,","venue":null,"work_id":"fe671fc7-bad8-49b6-8919-80808e2832c4","year":2024},"citing_paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T00:23:39.522256Z"},"links":{"citing_paper":"/paper/2506.14226"},"observation_digest":"sha256:9a45e24654187aa08d6aeffa6230dcba949da22e990e29b5e0bc50447680394f","observation_id":"10f541d2-dcb5-47fb-b289-9d0f53cbba39","resolution":{"observed_at":"2026-08-07T00:23:40.326165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.14226","last_updated":"2025-06-17T06:29:58Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T00:15:53.851971Z","submitted_at":"2025-06-17T06:29:58Z","title":"Investigation of Zero-shot Text-to-Speech Models for Enhancing Short-Utterance Speaker Verification"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":12,"verified_exact":1,"verified_fuzzy":19},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 1 inbound Pith citation observation for arXiv:2506.14226."}