{"as_of":"2026-08-15T11:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:45c67903278649a04e0b62dada78f33e53c450c9f923b1374c6cb41d29580969","coverage":[{"denominator":59,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":59,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T13:35:06.973371Z","state":"measured"},{"denominator":59,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":59,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.00561/citation-record","integrity":"/paper/2509.00561/integrity","json":"/paper/2509.00561/citation-record.json","paper":"/paper/2509.00561"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:16.917173Z","title":"Any-to-many voice conversion with location-relative sequence-to-sequence modeling,","venue":null,"work_id":"c8e248ee-b40d-4b0d-8145-7fe9bc9798d5","year":2021},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:00.901093Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:f8a63f0387f35b5572dcc6ac9c33ae8f8852816ccdf838eb75e4ed80f03beda7","observation_id":"27804854-9e9c-40a2-a0dc-268577259b48","resolution":{"observed_at":"2026-08-05T13:35:16.961825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:16.748241Z","title":"Face-driven zero- shot voice conversion with memory-based face-voice alignment,","venue":null,"work_id":"06de8968-73e4-47ff-b1da-61bd49415d64","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.001637Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:2b0481fc4cf2aad6a6818463548fd717f46f865d442305093455c9d4b7fc9937","observation_id":"a68a4635-4cc9-4cf1-a46b-cd4ad6348205","resolution":{"observed_at":"2026-08-05T13:35:16.806833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:16.636773Z","title":"Antifake: Using adversarial audio to prevent unauthorized speech synthesis,","venue":null,"work_id":"996b4efb-de18-4fe9-807a-838ebba674cd","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.067024Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:441a27c736b83fc27a22a1a1c35d7fc077776e01b72bbb6df4025925cdfaf074","observation_id":"865df751-4edb-47f8-84d0-4370867574f3","resolution":{"observed_at":"2026-08-05T13:35:16.687090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:16.477245Z","title":"Pmvc: Data augmentation-based prosody modeling for expressive voice conversion,","venue":null,"work_id":"a8f6a533-f307-4066-9151-e63bf1b8a94b","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.178662Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:87ee334743133cd4fda52bd66b5b18bdddce9e01b27dd198218268d3b3736907","observation_id":"d743fc0f-d45f-4993-b7db-fc4877e761eb","resolution":{"observed_at":"2026-08-05T13:35:16.541292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:16.339365Z","title":"Empathetic speech synthesis and testing for healthcare robots,","venue":null,"work_id":"11c8d729-119d-4ba3-9b65-688746c4ce6a","year":2021},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.256381Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:59df8f65066a713f657ed1ce5938f76bcb1746308eb151bc3aee15cf562f4541","observation_id":"c567c470-1f75-41ec-adc4-9cf1edae093e","resolution":{"observed_at":"2026-08-05T13:35:16.417798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.06037","last_updated":"2019-06-25T21:34:10Z","snapshot_observed_at":"2026-08-14T16:47:39.566745Z","submitted_at":"2019-04-12T05:15:31Z","title":"Direct speech-to-speech translation with a sequence-to-sequence model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.06037","snapshot_observed_at":"2026-08-05T13:35:01.357158Z","title":"Direct speech-to-speech translation with a sequence-to-sequence model (2019),","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.357158Z"},"links":{"cited_paper":"/paper/1904.06037","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:a84186ba653d2b415543637d8cccdfdfc8be789cf4b23d09451bf1dccd11238c","observation_id":"1e3e0b16-2f5b-4340-a23a-9ddea6ffc574","resolution":{"observed_at":"2026-08-05T13:35:01.357158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:16.150619Z","title":"Quality of life of patients after total laryngectomy: the struggle against stigmatization and social ex- clusion using speech synthesis,","venue":null,"work_id":"4d06a7e6-e660-4010-9a33-ce1e958c1d33","year":2018},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.396085Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:ed797bfe7d88257fcbf229133ae1b7cf925a3de27ca9b8e79e68597237d3d280","observation_id":"c3729c94-fff1-4ebf-bd0a-54d611b52b4e","resolution":{"observed_at":"2026-08-05T13:35:16.255425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:15.936461Z","title":"Overcoming the language barrier with speech translation technology,","venue":null,"work_id":"30436152-b6d8-4857-87df-8dc1cfd41fa2","year":2009},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.470542Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:289736b2edbd63b036c6728679c89c9c3e1f0f2150cf49f930c2b2ebc3348844","observation_id":"1d883b7a-5224-45c9-9025-3dc19ce2d5d0","resolution":{"observed_at":"2026-08-05T13:35:16.046900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:15.680444Z","title":"Adver- sarial speech for voice privacy protection from personalized speech generation,","venue":null,"work_id":"160feadd-44c4-4b11-a710-0a882dc6ef8d","year":2024},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.608324Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:ab4caebd822a2de8bad3f21c7bcd36966f07e413cc28046daad217fd9b46f865","observation_id":"ab7f624e-de8b-45d8-bc43-f144ff51b652","resolution":{"observed_at":"2026-08-05T13:35:15.807844Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17133","last_updated":"2024-12-01T04:06:27Z","snapshot_observed_at":"2026-08-13T12:10:25.350189Z","submitted_at":"2024-01-30T16:07:44Z","title":"SongBsAb: A Dual Prevention Approach against Singing Voice Conversion based Illegal Song Covers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17133","snapshot_observed_at":"2026-08-05T13:35:01.754281Z","title":"A proactive and dual prevention mechanism against illegal song covers empowered by singing voice conversion,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.754281Z"},"links":{"cited_paper":"/paper/2401.17133","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:2aa0da33028e5da7b5481eb562cc7ae68b1316f55f8bf181d087eca0e923b56e","observation_id":"bc60a99c-f76d-489a-a4d7-1ff95117813e","resolution":{"observed_at":"2026-08-05T13:35:01.754281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-15T09:40:37.844367Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-05T13:35:01.874077Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero- shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.874077Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:a8cde80e8f6b9b46480634ec6801181908a5065003d1f310b9fb3bd2f65f0602","observation_id":"74f964eb-a931-415f-8f63-f15b96a40155","resolution":{"observed_at":"2026-08-05T13:35:01.874077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:15.526342Z","title":"Generspeech: Towards style transfer for generalizable out-of-domain text-to-speech,","venue":null,"work_id":"719c7e90-b146-44e5-81d5-4b90ab9775cf","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.982576Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:d7026d4cc0edd9a8bc8ce96f5c89ccc4e0aaf0419412ff0ecf937dc7e50a7c27","observation_id":"634a1869-98b8-4b97-bab5-99a32a530a62","resolution":{"observed_at":"2026-08-05T13:35:15.619713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.09784","last_updated":"2023-05-22T16:09:26Z","snapshot_observed_at":"2026-08-13T15:40:31.062247Z","submitted_at":"2022-05-19T18:13:23Z","title":"End-to-End Zero-Shot Voice Conversion with Location-Variable Convolutions","version":3},"cited_work":{"arxiv_id":"2205.09784","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.09784","snapshot_observed_at":"2026-08-05T13:35:08.276641Z","title":"End-to-End Zero-Shot Voice Conversion with Location-Variable Convolutions","venue":"eess.AS","work_id":"f752eb7a-fea9-4082-b877-7ff452d758c1","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.093285Z"},"links":{"cited_paper":"/paper/2205.09784","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:76aed89ae8916a37650cf29e340707f2d8e7bf14f6e6fbe276fe24303f95c766","observation_id":"a4a1a4bf-e596-4c13-b17d-126a374fa9b2","resolution":{"observed_at":"2026-08-05T13:35:08.334352Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:15.319164Z","title":"Fraudsters used ai to mimic ceo’s voice in unusual cybercrime case,","venue":null,"work_id":"0bc0dd52-d1a6-4e62-b7cb-ea3982d35e35","year":2019},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.278790Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:12c236e8b412957d1f2f29ea064c8289333f5919485cb4c5b28d8c32db2b709d","observation_id":"a5ec7218-27ce-4601-8e2d-9527cd8e42b1","resolution":{"observed_at":"2026-08-05T13:35:15.400551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:15.097816Z","title":"Protecting your voice from speech synthesis attacks,","venue":null,"work_id":"b6835570-bb6b-4ec7-b907-22edc1508f73","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.357588Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:19ccba4f3f3450227cc7a93f2c93423bcb796014ede893a81e0b46d1cb5b842e","observation_id":"3719ef2a-27e1-4b6b-9b24-03a9d6db47f7","resolution":{"observed_at":"2026-08-05T13:35:15.204488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:14.900164Z","title":"V oice conversion based on maximum-likelihood estimation of spectral parameter trajectory,","venue":null,"work_id":"86c01962-e8ba-4690-9756-fd894b17523d","year":2007},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.418985Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:42568dd47f1e233ed6abd571b2ecd2b66d6e6202d1694a72e5d8653e2e1f4a12","observation_id":"edb15507-aa8f-49a8-8715-6707c239b4bc","resolution":{"observed_at":"2026-08-05T13:35:14.996996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:14.692246Z","title":"Robust processing techniques for voice conversion,","venue":null,"work_id":"7cc921e9-f867-4d49-988d-5c8b50b42ce3","year":2006},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.471900Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:799ccea34a538f82511967cecf87a5b4cd5b53b562885c2ea12f494ddceacf78","observation_id":"7be75a43-ee98-4469-ad08-268005c1b970","resolution":{"observed_at":"2026-08-05T13:35:14.796536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:14.523304Z","title":"V oice conversion using deep bidirectional long short-term memory based recurrent neural networks,","venue":null,"work_id":"16c1acb9-e89e-4e73-8d44-cfc494cd6af0","year":2015},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.582548Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:df02128065ee505a813b4979376626155cff3c90e5edc64d2e92d4674d710e98","observation_id":"c756487c-25cd-48dd-a064-78fc76a3ccf9","resolution":{"observed_at":"2026-08-05T13:35:14.590495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:14.393763Z","title":"V oice conversion with transformer network,","venue":null,"work_id":"0fdc1d3a-d0af-49c0-aaa2-3a166363d49e","year":2020},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.725736Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:00e00ea3eb44e82baec4413f6ed52de0ea5375124c4ad86bf40b464f68e7ce42","observation_id":"fa912839-a150-448d-8830-a38829d95ff8","resolution":{"observed_at":"2026-08-05T13:35:14.460469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:14.182577Z","title":"V oice conver- sion using partial least squares regression,","venue":null,"work_id":"80d35920-50a2-4827-9e3b-a82b4e19b6f6","year":2010},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.866349Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:a81c8969bbd25de6fe4e5d5be4e70777101fd479e5c57fef84797ff6f73db002","observation_id":"67d2146a-3dcf-4c16-941d-03d473cdd817","resolution":{"observed_at":"2026-08-05T13:35:14.277379Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:14.019027Z","title":"V oice conversion from non-parallel corpora using variational auto-encoder,","venue":null,"work_id":"1e7ca3b4-f6ac-434f-86a2-dbbad3b3bcde","year":2016},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:02.994794Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:0c457ae61874ecab6699d2abea1ce7bd375518350b72871bfa9ab06598349589","observation_id":"821530be-85de-4de8-8786-6332ec6f3db8","resolution":{"observed_at":"2026-08-05T13:35:14.099790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:13.799231Z","title":"One-shot voice conversion by vector quan- tization,","venue":null,"work_id":"3c42fef8-7e04-438f-bc35-bdf1c38180c0","year":2020},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:03.123846Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:2c8edbbf953adb95c695353a524c30ed511e85d1ccd22316da3150cd5b931618","observation_id":"f898e267-9225-4c66-9215-a4d501d5f9fc","resolution":{"observed_at":"2026-08-05T13:35:13.927011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:13.582721Z","title":"Non-parallel training in voice conversion using an adaptive restricted boltzmann machine,","venue":null,"work_id":"87e0723e-13cb-4bb9-9837-2a1d84d5a9cd","year":2032},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:03.232194Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:9b88d6a4d5b59254af2c0cfd92b6dd89f51e2e53fd1ac7d58e1bfa64c6be2f40","observation_id":"29c02670-57cf-44e3-91f7-44e260833982","resolution":{"observed_at":"2026-08-05T13:35:13.690758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:13.395007Z","title":"Non-parallel voice conversion using variational autoencoders conditioned by phonetic pos- teriorgrams and d-vectors,","venue":null,"work_id":"011d2d90-8dd7-4248-8810-20777ce97963","year":2018},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:03.350002Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:edc3f02c3b3f77d84f3dd8ae6b2d0282e04c5ee8e2f2a288ccdf16dc94c188d4","observation_id":"9beb8842-7aaf-4f90-af9d-354de317fbd9","resolution":{"observed_at":"2026-08-05T13:35:13.487782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:13.195760Z","title":"Again-vc: A one- shot voice conversion using activation guidance and adaptive instance normalization,","venue":null,"work_id":"cd39c38e-0561-4539-a751-f97b419da981","year":2021},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:03.467253Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:dce5dfca5eb863df338bfa079218ad9e5633b965d61c6672802ea43c5342eeaa","observation_id":"17fcd65c-0b06-4d55-a1a6-865a6d5908c8","resolution":{"observed_at":"2026-08-05T13:35:13.271106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:12.965789Z","title":"Nvc-net: End-to-end adversarial voice conversion,","venue":null,"work_id":"d6c1ec3c-361d-4989-8325-0a796b59091c","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:03.602638Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:351993e71aefdf76cae7397d63ef2fe3c1c6a986ef1040e395f4a5f8e642ee6d","observation_id":"f005ff6f-529e-42d4-9f08-b3b64c7fee67","resolution":{"observed_at":"2026-08-05T13:35:13.049287Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:12.775941Z","title":"Lm-vc: Zero-shot voice conversion via speech generation based on language models,","venue":null,"work_id":"c717bdcf-98d7-47fa-ad2c-5e192abacd34","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:03.812078Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:dec1ac618c4d9848531796ef59730fc574a6317760df60ef74d3ca2c4a23ddc0","observation_id":"4c9f637c-9439-4798-ad63-024d9b1a3a95","resolution":{"observed_at":"2026-08-05T13:35:12.864936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:12.600082Z","title":"Transfer learning from speaker verification to multispeaker text-to-speech synthesis,","venue":null,"work_id":"135caf43-9e3f-4d44-bf91-40c2da437486","year":2018},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:03.927168Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:0240cafbfd2c6256f6f02983313afae9e219fc756ec4114c438f0678c11df91b","observation_id":"07cda1d7-ce9f-4385-ae97-fa648abb1f45","resolution":{"observed_at":"2026-08-05T13:35:12.718366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:04.011922Z","title":"Fastspeech: Fast, robust and controllable text to speech,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.011922Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:9fb79e3b537c91c7a49f8126e8ae69693c29a446ebc74e989b0532a844ddd05d","observation_id":"d84f5f34-3278-44ad-8261-ccb0dbc890d0","resolution":{"observed_at":"2026-08-05T13:35:04.011922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:12.369054Z","title":"Neural text-to-speech adaptation from low quality public recordings,","venue":null,"work_id":"3d51c4d5-6c15-4670-8dd7-3556aa9b7deb","year":2019},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.134337Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:cb3a6306b62daa0db8268cf35778d7bf6ad903491ddaca29448605b2d7e6a4fa","observation_id":"c67f4af1-3864-44d9-9423-3621934fff82","resolution":{"observed_at":"2026-08-05T13:35:12.478350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-05T13:35:04.280612Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.280612Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:bda5b47cb76cc6ff4e8dc0f04c3a6bc1f262110a410bf8139ee5b0b0b1c33609","observation_id":"caa4fed2-b6f6-4b3a-bfee-25ed8bfb1ed6","resolution":{"observed_at":"2026-08-05T13:35:04.280612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:12.169518Z","title":"V oid: A fast and light voice liveness detection system,","venue":null,"work_id":"4919ecaf-ec21-46a1-ab4d-0a685cfc8d59","year":2020},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.374439Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:24f9c6a05eeab0a87e896c8c836b320da2fbb2b5db2e1794ae884084e42954af","observation_id":"9a9e0fe4-2852-4543-9632-d3396f4f6d45","resolution":{"observed_at":"2026-08-05T13:35:12.289123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:11.941116Z","title":"Detecting ai-synthesized speech using bispectral analysis","venue":null,"work_id":"cab34cda-13ae-4739-a28a-2c41a8d88a14","year":2019},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.433551Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:bc468c0f3eba539494d75d4115eca31788937d84f80f8d04c096b903fc8b9464","observation_id":"12e25d4b-1572-4aae-9df3-afb609149bb0","resolution":{"observed_at":"2026-08-05T13:35:12.028697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:11.661889Z","title":"Who are you (i really wanna know)? detecting audio {DeepFakes} through vocal tract reconstruction,","venue":null,"work_id":"f86d0c69-a534-4fa5-b39a-8b5419eda5ed","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.539265Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:abd00c072face7778cf90f4f2d3b8d519fa856732b99a7a62dd016664152d61a","observation_id":"7b4c0657-2f83-4002-8894-87553f49f9d3","resolution":{"observed_at":"2026-08-05T13:35:11.784925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03410","last_updated":"2023-12-06T10:48:36Z","snapshot_observed_at":"2026-08-13T05:08:51.509631Z","submitted_at":"2023-12-06T10:48:36Z","title":"Detecting Voice Cloning Attacks via Timbre Watermarking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03410","snapshot_observed_at":"2026-08-05T13:35:04.611485Z","title":"De- tecting voice cloning attacks via timbre watermarking,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.611485Z"},"links":{"cited_paper":"/paper/2312.03410","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:63c06e28cc2a6e6e7ef819713b18eb751c0c628adaf1fa3ac483013198c2fa18","observation_id":"b25f9637-0742-4c1f-8319-cf6661a2d260","resolution":{"observed_at":"2026-08-05T13:35:04.611485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.15655","last_updated":"2024-05-27T02:33:22Z","snapshot_observed_at":"2026-08-13T21:17:25.842655Z","submitted_at":"2024-05-24T15:49:00Z","title":"HiddenSpeaker: Generate Imperceptible Unlearnable Audios for Speaker Verification System","version":2},"cited_work":{"arxiv_id":"2405.15655","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.15655","snapshot_observed_at":"2026-08-05T13:35:08.116608Z","title":"HiddenSpeaker: Generate Imperceptible Unlearnable Audios for Speaker Verification System","venue":"cs.SD","work_id":"6c869124-30d1-4121-a07a-c7be637f503d","year":2024},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.731874Z"},"links":{"cited_paper":"/paper/2405.15655","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:025323834f3f6018d92448f085cbdc0c14104b5ef6668d8b6f38373f9b3868d8","observation_id":"7b39d6bf-0f3c-4ce2-89b2-4e3d2a525354","resolution":{"observed_at":"2026-08-05T13:35:08.177520Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:11.459896Z","title":"Defending your voice: Adversarial attack on voice conversion,","venue":null,"work_id":"b80fbaeb-0089-406a-935c-0bc4f502feb8","year":2021},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.848118Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:53bea8248d60d7d98489469d970c505da2a92d774f9e262d30d80e5abcd9d072","observation_id":"5c2053ec-0dd6-4ec4-9a03-7f35fd278384","resolution":{"observed_at":"2026-08-05T13:35:11.563644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.12932","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:07.954180Z","title":"Enkidu: Universal frequential perturbation for real-time audio privacy protection against voice deepfakes,","venue":null,"work_id":"e420f2aa-f144-4ce9-8a76-38caea649b8a","year":2025},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:04.976629Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:0d4b67c1ef38b2f9aad7bc4725753c86bc17222a56a80f90e8cbfb9856d01a0e","observation_id":"3967871c-7831-4d1c-8e9b-80e803b9e384","resolution":{"observed_at":"2026-08-05T13:35:08.040332Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:11.206865Z","title":"V oice guard: Protecting voice privacy with strong and imperceptible adversarial perturbation in the time domain","venue":null,"work_id":"f330e7d5-17a4-4c95-b64d-ddbce8661117","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.096850Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:7b30a333afd49dc4d1328e32cc028241595041d23d519c5ef66c80407c91c862","observation_id":"36cdcf47-5802-4652-be4e-1b551f17ff81","resolution":{"observed_at":"2026-08-05T13:35:11.317979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:10.918672Z","title":"Active defense against voice conversion through generative adversarial network,","venue":null,"work_id":"610d3bfb-b309-4e3f-84ec-4a08dc8c3e25","year":2024},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.207623Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:388b551b658e8dbc15910ddaf5422a191bf95517fcae2a5d99a3b8d2de12b512","observation_id":"663cd7fa-a66e-4511-8fcb-f5d4f9978b42","resolution":{"observed_at":"2026-08-05T13:35:11.065129Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.13497","last_updated":"2022-03-25T08:14:37Z","snapshot_observed_at":"2026-08-14T23:03:39.728900Z","submitted_at":"2022-03-25T08:14:37Z","title":"WaveFuzz: A Clean-Label Poisoning Attack to Protect Your Voice","version":1},"cited_work":{"arxiv_id":"2203.13497","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.13497","snapshot_observed_at":"2026-08-05T13:35:07.590982Z","title":"WaveFuzz: A Clean-Label Poisoning Attack to Protect Your Voice","venue":"cs.SD","work_id":"dfc01ab3-8340-4823-ab98-638d8c480c86","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.327401Z"},"links":{"cited_paper":"/paper/2203.13497","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:745e94e21be80e5aa4c94f2bb08842e19e69b628f520017297950f528ae4e8cc","observation_id":"6ca136de-c169-480c-b6f8-b79003af8cd4","resolution":{"observed_at":"2026-08-05T13:35:07.697684Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:10.607810Z","title":"https://www.bbc.com/news/technology-60780142,","venue":null,"work_id":"a75fd0bb-c6ea-404c-93c2-042729dd1e4e","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.422111Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:e68806c9c926d7a86375b5069d7eda28c4d62532a1057a0317a2a22d515a3de6","observation_id":"f5eccf13-cc85-472a-ab34-2d4ec80029cc","resolution":{"observed_at":"2026-08-05T13:35:10.765319Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15087","last_updated":"2023-09-26T17:31:35Z","snapshot_observed_at":"2026-08-13T10:04:31.673580Z","submitted_at":"2023-09-26T17:31:35Z","title":"Privacy-preserving and Privacy-attacking Approaches for Speech and Audio -- A Survey","version":1},"cited_work":{"arxiv_id":"2309.15087","doi":null,"metadata_source":"pith","pith_arxiv_id":"2309.15087","snapshot_observed_at":"2026-08-05T13:35:07.428374Z","title":"Privacy-preserving and Privacy-attacking Approaches for Speech and Audio -- A Survey","venue":"cs.CR","work_id":"10f54d19-f5ca-4c9a-84ec-db83e7d7de72","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.497799Z"},"links":{"cited_paper":"/paper/2309.15087","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:03fc6408b3354ae4d9e0019824d7300eeea4249a4be9238964bdf28dcf50b3b4","observation_id":"112ffa83-12cb-41d0-ada0-3909f808f82e","resolution":{"observed_at":"2026-08-05T13:35:07.492065Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:10.330752Z","title":"Personal voice assistant security and pri- vacy—a survey,","venue":null,"work_id":"f06bc033-7a8e-466e-ab4e-9b21bd745414","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.613549Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:e05460e5bd900d0cd72ffbc61deee961e53a0019682ff86eb71eb0a290464904","observation_id":"7de4ce88-3c75-42f0-a8c6-cfa5ee161674","resolution":{"observed_at":"2026-08-05T13:35:10.461737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:10.098945Z","title":"Yourtts: Towards zero-shot multi-speaker tts and zero-shot voice conversion for everyone,","venue":null,"work_id":"a4a15efa-5850-4066-a352-25bffc177cef","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.729851Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:4937628f84bb13f8ce54ad563b85b04ccc35c06b1b50bf1488bf3765ba10bbea","observation_id":"5ad00950-94a8-47d8-9f0c-f4ff227c219a","resolution":{"observed_at":"2026-08-05T13:35:10.221375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:05.851154Z","title":"Glow-tts: A generative flow for text-to-speech via monotonic alignment search,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.851154Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:1790e82933fd92c59114421209fc3cd63b5273cbb9346e1af281fa090a1d28fd","observation_id":"0ed5bf4c-67c5-447e-abef-55114ae017a4","resolution":{"observed_at":"2026-08-05T13:35:05.851154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:09.830550Z","title":"Tacotron: Towards end-to-end speech synthesis,","venue":null,"work_id":"3b66687b-88c1-4204-8d9b-b8ed5acd54f5","year":2017},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:05.929475Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:e7e4a964f6187a2d9c2848f6c8d55da9a2624d08d7aa547e09a0107ceb638fee","observation_id":"5e899f88-9872-4874-b874-e661cdffd3f3","resolution":{"observed_at":"2026-08-05T13:35:09.953356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:09.599789Z","title":"Boosting adversarial transferability by achieving flat local maxima,","venue":null,"work_id":"fb45ee3e-ebdd-4e9e-ac4e-930e21ab3be9","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.027313Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:eaff07897381aa3eddc7671143728b701e83cd5f38b8066fb21e29e9f872d988","observation_id":"ca7e5cee-816a-4624-bd05-438c1f66915e","resolution":{"observed_at":"2026-08-05T13:35:09.702599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:06.086284Z","title":"Librispeech: an asr corpus based on public domain audio books,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.086284Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:590ebd16ce91a4cbf0d622024c58379bc703dc1819b5ca99bb7c7d1840d17339","observation_id":"2e7895eb-d9e5-451f-bc6e-60778cfacff7","resolution":{"observed_at":"2026-08-05T13:35:06.086284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.08612","last_updated":"2018-05-30T06:52:06Z","snapshot_observed_at":"2026-08-14T20:51:29.635557Z","submitted_at":"2017-06-26T21:42:27Z","title":"VoxCeleb: a large-scale speaker identification dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.08612","snapshot_observed_at":"2026-08-05T13:35:06.215738Z","title":"V oxceleb: a large-scale speaker identification dataset,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.215738Z"},"links":{"cited_paper":"/paper/1706.08612","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:b97364c7e123f62e828317874f42f0446b8c4cf2689359f7cef2af0e8dcb6d5f","observation_id":"9b07244d-f2ec-4073-aab3-026620e102e1","resolution":{"observed_at":"2026-08-05T13:35:06.215738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:09.328080Z","title":"https://github.com/coqui-ai/tts,","venue":null,"work_id":"103ae33f-4e9d-4a2d-a4fd-13846ad02f77","year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.286535Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:8b2975b88a51483bda08a4f061168ff1fd1c0bb35784d19d24fe69bb19ad8047","observation_id":"73dcddd4-5945-458d-a079-c37a59cab544","resolution":{"observed_at":"2026-08-05T13:35:09.467999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.10375","last_updated":"2022-01-26T12:30:12Z","snapshot_observed_at":"2026-08-13T16:53:31.434751Z","submitted_at":"2022-01-25T15:06:07Z","title":"Zero-Shot Long-Form Voice Cloning with Dynamic Convolution Attention","version":2},"cited_work":{"arxiv_id":"2201.10375","doi":null,"metadata_source":"pith","pith_arxiv_id":"2201.10375","snapshot_observed_at":"2026-08-05T13:35:07.242747Z","title":"Zero-Shot Long-Form Voice Cloning with Dynamic Convolution Attention","venue":"eess.AS","work_id":"e2be4e7d-ab05-4344-882c-56e7c5546f01","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.355783Z"},"links":{"cited_paper":"/paper/2201.10375","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:eef93895a7d89d14cf5028e8a4e1861021c0382993a0d0ff26adb65e585e2b9f","observation_id":"3c539e35-1fa9-40fa-a95f-2e9c7f6e57ab","resolution":{"observed_at":"2026-08-05T13:35:07.330941Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2008.03802","last_updated":"2020-08-09T20:00:57Z","snapshot_observed_at":"2026-08-13T21:54:09.948911Z","submitted_at":"2020-08-09T20:00:57Z","title":"SpeedySpeech: Efficient Neural Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2008.03802","doi":null,"metadata_source":"pith","pith_arxiv_id":"2008.03802","snapshot_observed_at":"2026-08-05T13:35:07.085963Z","title":"SpeedySpeech: Efficient Neural Speech Synthesis","venue":"eess.AS","work_id":"6bd22d43-aac2-4682-95d2-3762f9fdd217","year":2020},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.412399Z"},"links":{"cited_paper":"/paper/2008.03802","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:e78587b5ea0a380dad65da8ec06ab4edf8e4ad9f245db2e9ceb484bfed9cb271","observation_id":"f5ee70f6-27b2-4e98-ba88-c26600c023bc","resolution":{"observed_at":"2026-08-05T13:35:07.152077Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:06.495975Z","title":"Natural tts synthesis by conditioning wavenet on mel spectrogram predictions,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.495975Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:a8e3181c8b606d5b71a0645fc1b8f8a40b6d78b2c7c002d485dc16dad15c3a0a","observation_id":"00e993f5-eb35-4246-8a93-0805a7a23582","resolution":{"observed_at":"2026-08-05T13:35:06.495975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:09.090610Z","title":"Speech- brain: A general-purpose speech toolkit,","venue":null,"work_id":"6f9c2d61-b74a-4ec0-9d23-198467ac7551","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.581021Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:e5465c76dd95c39ad3f2b89b541326e8f0d49234ca6713b3b49a8948aaa229e7","observation_id":"c7d9c461-0d08-498f-b0de-61e5716b9f42","resolution":{"observed_at":"2026-08-05T13:35:09.191711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:08.877235Z","title":"https://github.com/speechbrain/speechbrain,","venue":null,"work_id":"a10e601a-15bd-4b03-a915-4d8f84716685","year":2024},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.718398Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:bbf3787b947329c8df8bbd30f202e6c390c7295bb7f1404bbff7fe0147699b75","observation_id":"dfa9b7a1-3ea5-4a20-86e5-f3df4685ac86","resolution":{"observed_at":"2026-08-05T13:35:08.958274Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19971","last_updated":"2024-12-27T02:48:43Z","snapshot_observed_at":"2026-08-13T04:56:21.956002Z","submitted_at":"2024-03-29T04:42:12Z","title":"3D-Speaker-Toolkit: An Open-Source Toolkit for Multimodal Speaker Verification and Diarization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19971","snapshot_observed_at":"2026-08-05T13:35:06.838387Z","title":"3d-speaker-toolkit: An open source toolkit for multi-modal speaker verification and diarization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.838387Z"},"links":{"cited_paper":"/paper/2403.19971","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:1e8cbf7458da556f3d2e408f8ac441f6233f70e2b65828e7059babd51faca090","observation_id":"1ce33aae-313a-404b-a193-d5144b6598ce","resolution":{"observed_at":"2026-08-05T13:35:06.838387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:08.683830Z","title":"https://github.com/modelscope/3d-speaker,","venue":null,"work_id":"bfbf122b-9ecb-4e6d-aa92-7e0a7fc35a33","year":2024},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.917849Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:8d1029405baa6f1f6e4f51eb109caeac2ebd861ad7d85ae5314e710374592155","observation_id":"1aeaf627-1e6e-4a6a-9c8f-33df1b80a6b4","resolution":{"observed_at":"2026-08-05T13:35:08.764982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:35:08.493181Z","title":"https://openai.com/index/whisper/,","venue":null,"work_id":"c492a497-600b-43ac-9a50-5fb0174bb39d","year":2022},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:06.973371Z"},"links":{"citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:2db2d0002c8b22400e6873be4792489a1bae09bdefb878b814cedae06fc4fcdc","observation_id":"e1b074ce-b8d8-4df7-8a80-7af0efb3503a","resolution":{"observed_at":"2026-08-05T13:35:08.578320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-13T06:14:36.981527Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks"},"reference_resolution":{"displayed":59,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":7,"verified_fuzzy":41},"total_outbound_references":59},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 59 of 59 outbound references and 0 inbound Pith citation observations for arXiv:2509.00561."}