{"as_of":"2026-08-12T06:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fff330c5b3b27ba9f1712f3e82b61b0be785c18994b2bf0b3eaf266d9d315d1a","coverage":[{"denominator":97,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":97,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:34:37.749011Z","state":"measured"},{"denominator":107,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":107,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:43:08.326154Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T07:49:38.750008Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":"2501.08238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-04T07:49:38.750008Z","title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","venue":"cs.SD","work_id":"da148f0e-02bf-4b02-a1d3-d611d870a9ae","year":2025},"citing_paper":{"arxiv_id":"2504.08528","last_updated":"2026-04-07T06:11:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-11T13:40:53Z","title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T20:44:57.476464Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2504.08528"},"observation_digest":"sha256:aa0fb57f2968321f0f5b2893da84270b34b2ed23b3009e2f41d3b2b71cd47175","observation_id":"88ac146d-6375-4b72-bb2a-8241fc8b30fe","resolution":{"observed_at":"2026-06-09T02:05:18.403807Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-08-07T05:43:08.326154Z","title":"Codec- Fake+: A large-scale neural audio codec-based deepfake speech dataset,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07294","last_updated":"2025-08-16T22:33:43Z","snapshot_observed_at":"2026-08-11T16:22:46.558963Z","submitted_at":"2025-06-08T21:36:10Z","title":"Towards Generalized Source Tracing for Codec-Based Deepfake Speech","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:43:08.326154Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2506.07294"},"observation_digest":"sha256:b3e460de210bf687d7e962a400e88410ff6072f3d901b34a2c2ec6d70617bac1","observation_id":"e1dcd91c-70e6-4704-ad9d-b2be91fa50df","resolution":{"observed_at":"2026-08-07T05:43:08.326154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-08-05T21:11:12.047177Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09294","last_updated":"2025-08-12T19:15:13Z","snapshot_observed_at":"2026-08-09T19:48:25.235783Z","submitted_at":"2025-08-12T19:15:13Z","title":"Fake-Mamba: Real-Time Speech Deepfake Detection Using Bidirectional Mamba as Self-Attention's Alternative","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T21:11:12.047177Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2508.09294"},"observation_digest":"sha256:852aa3b3f6bb7b0932a0ec0107d39b1ae8db54889f112a77a315f0240d413376","observation_id":"2039c206-32a9-4229-a07e-a06ee903e5b9","resolution":{"observed_at":"2026-08-05T21:11:12.047177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":"2501.08238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-04T07:49:38.750008Z","title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","venue":"cs.SD","work_id":"da148f0e-02bf-4b02-a1d3-d611d870a9ae","year":2025},"citing_paper":{"arxiv_id":"2603.05373","last_updated":"2026-04-11T06:15:04Z","snapshot_observed_at":"2026-08-11T17:22:26.053012Z","submitted_at":"2026-03-05T16:59:26Z","title":"Hierarchical Decoding for Discrete Speech Synthesis with Multi-Resolution Spoof Detection","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T15:07:27.418534Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2603.05373"},"observation_digest":"sha256:60534e82359fbfcb17e558a34dae335e97edc4aa72e98cf123ac66dc3efb1b3e","observation_id":"1b2a14ab-d35f-4de7-9ea5-9f25cd65f274","resolution":{"observed_at":"2026-06-09T02:05:18.403807Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":"2501.08238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-04T07:49:38.750008Z","title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","venue":"cs.SD","work_id":"da148f0e-02bf-4b02-a1d3-d611d870a9ae","year":2025},"citing_paper":{"arxiv_id":"2604.17642","last_updated":"2026-04-19T22:26:28Z","snapshot_observed_at":"2026-08-10T22:40:30.639788Z","submitted_at":"2026-04-19T22:26:28Z","title":"HCFD: A Benchmark for Audio Deepfake Detection in Healthcare","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-10T04:52:52.682483Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2604.17642"},"observation_digest":"sha256:7bd7bb6de823f5203b7ed62a3dbe8721e61b6507a00ef26bfc01d9a66b598092","observation_id":"1ffcbd3f-449e-49e0-b900-f4c854354782","resolution":{"observed_at":"2026-06-09T02:05:18.403807Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":"2501.08238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-04T07:49:38.750008Z","title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","venue":"cs.SD","work_id":"da148f0e-02bf-4b02-a1d3-d611d870a9ae","year":2025},"citing_paper":{"arxiv_id":"2604.19949","last_updated":"2026-04-21T19:54:54Z","snapshot_observed_at":"2026-07-06T23:06:31.110859Z","submitted_at":"2026-04-21T19:54:54Z","title":"Indic-CodecFake meets SATYAM: Towards Detecting Neural Audio Codec Synthesized Speech Deepfakes in Indic Languages","version":1},"reference_index":175,"source":"arxiv_source","source_observed_at":"2026-05-10T00:34:30.978387Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2604.19949"},"observation_digest":"sha256:d796296d82caf0ec78e1c7711d3ed151c1f9f425049d0a59f594c4d53a182164","observation_id":"39430e9b-6683-416e-a6c8-ecc49f40dae6","resolution":{"observed_at":"2026-06-09T02:05:18.403807Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":"2501.08238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-04T07:49:38.750008Z","title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","venue":"cs.SD","work_id":"da148f0e-02bf-4b02-a1d3-d611d870a9ae","year":2025},"citing_paper":{"arxiv_id":"2606.07494","last_updated":"2026-06-05T17:48:46Z","snapshot_observed_at":"2026-08-03T05:11:42.143646Z","submitted_at":"2026-06-05T17:48:46Z","title":"Mitigating Proxy-to-Wild Domain Gap in Deepfake Speech","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T20:43:29.213147Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2606.07494"},"observation_digest":"sha256:08eebc2492fb966f21641cac4bb408d7f0b68572f55d1f45dc5cfac9a0267145","observation_id":"a2d4043f-028c-49fb-a413-b95aa6477278","resolution":{"observed_at":"2026-07-02T20:17:21.992012Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":"2501.08238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-04T07:49:38.750008Z","title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","venue":"cs.SD","work_id":"da148f0e-02bf-4b02-a1d3-d611d870a9ae","year":2025},"citing_paper":{"arxiv_id":"2606.21735","last_updated":"2026-06-19T20:45:55Z","snapshot_observed_at":"2026-07-06T23:56:40.795364Z","submitted_at":"2026-06-19T20:45:55Z","title":"Bridging the Age Gap: Towards Detecting Neural Audio Codec Synthesized Elderly Speech Deepfake","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-26T12:51:14.919171Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2606.21735"},"observation_digest":"sha256:c7bda390297f1400850da5605dcfa89c108ef4e5caf90e62565540b8517051de","observation_id":"8c9a885c-8172-4f02-bfab-83c651006246","resolution":{"observed_at":"2026-07-04T07:49:38.751699Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":"2501.08238","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-04T07:49:38.750008Z","title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","venue":"cs.SD","work_id":"da148f0e-02bf-4b02-a1d3-d611d870a9ae","year":2025},"citing_paper":{"arxiv_id":"2607.00387","last_updated":"2026-07-01T03:32:08Z","snapshot_observed_at":"2026-08-01T10:46:33.437238Z","submitted_at":"2026-07-01T03:32:08Z","title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","version":1},"reference_index":115,"source":"pdf_text","source_observed_at":"2026-07-02T05:52:55.818877Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2607.00387"},"observation_digest":"sha256:5b27252a3a8cb781ee5b0fe96bff15831628e4efa446e33e484a237b3ec31462","observation_id":"d25d6f4b-a15e-43e2-af56-cd3eb9fa178d","resolution":{"observed_at":"2026-07-02T05:56:39.919902Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08238","snapshot_observed_at":"2026-07-12T04:39:19.585879Z","title":"Codecfake+: A large-scale neural audio codec-based deepfake speech dataset,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.03134","last_updated":"2026-07-03T09:23:36Z","snapshot_observed_at":"2026-08-09T18:24:17.075944Z","submitted_at":"2026-07-03T09:23:36Z","title":"Open-Set Source Tracing as Compositional Factors via Structured Prototypes","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-12T04:39:19.585879Z"},"links":{"cited_paper":"/paper/2501.08238","citing_paper":"/paper/2607.03134"},"observation_digest":"sha256:36b53a09680075f634e696f96c16df01a40befa13335e44fb295993317d48263","observation_id":"6a429d88-5668-485e-91de-3a2666823a6e","resolution":{"observed_at":"2026-07-12T04:39:19.585879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.08238/citation-record","integrity":"/paper/2501.08238/integrity","json":"/paper/2501.08238/citation-record.json","paper":"/paper/2501.08238"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.267646Z","title":"CodecFake: Enhancing anti-spoofing models against deepfake audios from codec-based speech synthesis systems,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.267646Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:fc4b79486ef080dad702bbdc2176c3dc2bce004de571d9026805a8840a48447b","observation_id":"d4b578d3-d874-4d1f-a290-ff707811ede0","resolution":{"observed_at":"2026-08-10T20:34:37.267646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.273720Z","title":"A survey on speech deepfake detection,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.273720Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:14015c6885922a830922c69750fd3943eef953ec288be1460259d0acdbc41c52","observation_id":"d9d2fc84-9d57-4fe2-af52-37ed895f30a9","resolution":{"observed_at":"2026-08-10T20:34:37.273720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.278484Z","title":"Recurrent convolutional structures for audio spoof and video deepfake detection,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.278484Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:e0dd48b23a1b3ee02c81058d67b10df9b447f8f3d09b83712a09b3c39860404c","observation_id":"d18a1758-3bae-4953-8905-1b0ff38295ae","resolution":{"observed_at":"2026-08-10T20:34:37.278484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.284501Z","title":"Deep learning in face synthesis: A survey on deepfakes,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.284501Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:b42a9e00e427796d35c5a66209c1240944957412616f376d90a964602674345f","observation_id":"6c34585c-5571-452a-8116-acf49ff87de2","resolution":{"observed_at":"2026-08-10T20:34:37.284501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.290129Z","title":"The de- fender’s perspective on automatic speaker verification: An overview,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.290129Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:b5c193615f71b6274007efa7c7b2771375b87e154746700ab69697058e8bac2e","observation_id":"576b9393-9b51-4a99-9251-29419fe130c3","resolution":{"observed_at":"2026-08-10T20:34:37.290129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.295245Z","title":"ASVspoof 2015: the first automatic speaker verification spoofing and countermeasures challenge,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.295245Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:63a040e14e7a8c1266831fdef3276b31a74e17d494d67378b2690190a1777975","observation_id":"337c0e97-393a-4c82-95dd-948bba635f3f","resolution":{"observed_at":"2026-08-10T20:34:37.295245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.301316Z","title":"The ASVspoof 2017 challenge: Assessing the limits of replay spoofing attack detection,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.301316Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:27d9f97e802c8fbf5755db8f855828034dc28754083ee8a6496fc8212e46d1e1","observation_id":"99542994-0052-4001-b3f0-9d580d92530f","resolution":{"observed_at":"2026-08-10T20:34:37.301316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.306219Z","title":"ASVspoof 2019: Future horizons in spoofed and fake audio detection,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.306219Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:5957b242e6c7cfc7f64eeedb4d7ea23d60f72a3097dff16fdfa5ce3849f189ec","observation_id":"a83a549d-e0ed-4e27-82b9-da62c3db560e","resolution":{"observed_at":"2026-08-10T20:34:37.306219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.310983Z","title":"ASVspoof 2019: spoofing countermeasures for the detection of synthesized, converted and replayed speech,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.310983Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:3395a0541e49d40adc2c7fbaaedef2227654306aeb3ac077cf1d9936895b3979","observation_id":"0a1e7493-515b-489e-b104-88c7644e577b","resolution":{"observed_at":"2026-08-10T20:34:37.310983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.316832Z","title":"Add 2022: the first audio deep synthesis detection challenge,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.316832Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:10d3110c3db032659edcbba4308831d6a3421288c79d333544bca9b4d8fe6b2f","observation_id":"1c478cd5-f5d7-42e0-9819-2f9ff6a433c1","resolution":{"observed_at":"2026-08-10T20:34:37.316832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.324912Z","title":"Add 2023: the second audio deepfake detection challenge,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.324912Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:fe00ef65dbfaac809ebb0b9bed1c3222771c1a973305204709727bbf0741d730","observation_id":"8280eb71-d9a1-4fed-85a7-c63714b30bde","resolution":{"observed_at":"2026-08-10T20:34:37.324912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.330076Z","title":"The attacker’s perspective on automatic speaker verification: An overview,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.330076Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:04f6c7e6bc68bf33bc7c70f610276376ed58bc3ddcf5ca489775219777934789","observation_id":"e24c38ee-15da-4dcd-862d-2da27ad22739","resolution":{"observed_at":"2026-08-10T20:34:37.330076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.334452Z","title":"Soundstream: An end-to-end neural audio codec,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.334452Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:1efd49b0c238e6a0e5530104792c10f5aafb4d23d961b48bb620b1ec1ba7206c","observation_id":"58ca4a0c-8d56-4fd2-b493-df6b89af7689","resolution":{"observed_at":"2026-08-10T20:34:37.334452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.338617Z","title":"Harp-net: Hyper-autoencoded reconstruction propagation for scalable neural audio coding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.338617Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:e89610c25037965451bcedff5df06986924356fd5e2f0345512b1e59a2826772","observation_id":"390ca0ed-94d7-48d2-9848-1bb5ca33deb5","resolution":{"observed_at":"2026-08-10T20:34:37.338617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.342965Z","title":"End-to-end neural speech coding for real-time communications,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.342965Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:2d3d11fb87806130aa4f57c81cb290bbceb2f688c686bd1ed5ffb81925ff5c3c","observation_id":"d0ce5434-0539-4aee-aea3-ff26385c236f","resolution":{"observed_at":"2026-08-10T20:34:37.342965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.346994Z","title":"Cross-scale vector quantization for scalable neural speech coding,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.346994Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:acc0a580a5fa9c5ce5a19882ee4c47b3dbe3e17bcd343e963f869bfaf87e3ea3","observation_id":"9742f7f8-9fc9-40c6-8cd9-d7cf314af248","resolution":{"observed_at":"2026-08-10T20:34:37.346994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.351303Z","title":"Disentangled feature learning for real-time neural speech coding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.351303Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:f8576ca2ee5311edaf86cfb1b4384eba47e204d007a216528d07e1a27d126367","observation_id":"3b3b4f75-55a4-425c-940f-c79ee69d29c9","resolution":{"observed_at":"2026-08-10T20:34:37.351303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07243","last_updated":"2023-05-23T21:41:54Z","snapshot_observed_at":"2026-08-10T19:02:23.351550Z","submitted_at":"2023-05-12T04:19:49Z","title":"Better speech synthesis through scaling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07243","snapshot_observed_at":"2026-08-10T20:34:37.355515Z","title":"Better speech synthesis through scaling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.355515Z"},"links":{"cited_paper":"/paper/2305.07243","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:f9fff54469d9553fcf351159ffef306256bd67119e6836a5c023cb745d70450c","observation_id":"adb8d539-24ac-4398-adfd-285723d5dbdb","resolution":{"observed_at":"2026-08-10T20:34:37.355515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.09636","last_updated":"2023-05-16T17:41:25Z","snapshot_observed_at":"2026-08-06T19:45:49.830925Z","submitted_at":"2023-05-16T17:41:25Z","title":"SoundStorm: Efficient Parallel Audio Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.09636","snapshot_observed_at":"2026-08-10T20:34:37.360088Z","title":"Soundstorm: Efficient parallel audio generation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.360088Z"},"links":{"cited_paper":"/paper/2305.09636","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:e52b95a795cdcabd96c525cea817111a0ed901c87356b2a14e96830db8ac860a","observation_id":"00494c4c-04ba-4edb-882d-e3118ef4de32","resolution":{"observed_at":"2026-08-10T20:34:37.360088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00169","last_updated":"2024-07-22T09:53:44Z","snapshot_observed_at":"2026-08-11T17:03:08.296897Z","submitted_at":"2023-08-31T23:26:10Z","title":"RepCodec: A Speech Representation Codec for Speech Tokenization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00169","snapshot_observed_at":"2026-08-10T20:34:37.364442Z","title":"Repcodec: A speech representation codec for speech tokenization,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.364442Z"},"links":{"cited_paper":"/paper/2309.00169","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:616623fa02253275660bda4800446d6898fdd43d16aaa44a0279624384cb1615","observation_id":"d9d51869-dbf0-47ab-8c7f-a1c6857019a1","resolution":{"observed_at":"2026-08-10T20:34:37.364442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08093","last_updated":"2024-02-15T18:57:26Z","snapshot_observed_at":"2026-08-10T16:15:25.311057Z","submitted_at":"2024-02-12T22:21:30Z","title":"BASE TTS: Lessons from building a billion-parameter Text-to-Speech model on 100K hours of data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08093","snapshot_observed_at":"2026-08-10T20:34:37.370223Z","title":"Base tts: Lessons from building a billion-parameter text-to-speech model on 100k hours of data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.370223Z"},"links":{"cited_paper":"/paper/2402.08093","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:3aca3529c58865ae08a85839b130e0b6c27a0fda8bfac943ec5d3ab0173cccb5","observation_id":"c9b3b6e4-8b9e-4a4a-9458-19662c08bbd3","resolution":{"observed_at":"2026-08-10T20:34:37.370223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10533","last_updated":"2024-09-24T01:14:07Z","snapshot_observed_at":"2026-07-06T17:31:04.675896Z","submitted_at":"2024-02-16T09:38:16Z","title":"APCodec: A Neural Audio Codec with Parallel Amplitude and Phase Spectrum Encoding and Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10533","snapshot_observed_at":"2026-08-10T20:34:37.377168Z","title":"Apcodec: A neural audio codec with parallel amplitude and phase spectrum encoding and decoding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.377168Z"},"links":{"cited_paper":"/paper/2402.10533","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:2aa89e922d5cfb9a4cfd1a7ed4fc1b88dcecfeba50cf59f8e4dc3930dd169b95","observation_id":"77bf54ba-7d3a-4e59-bd0a-26afcf03ef29","resolution":{"observed_at":"2026-08-10T20:34:37.377168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.178462Z","title":"Esc: Efficient speech coding with cross-scale residual vector quantized transformers,","venue":null,"work_id":"b0830707-2f6a-4da7-8d8c-f38ce99557c8","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.382855Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:8f116586ffcb7f44d3780a7ad393ee101c0b7acdccd6036eff0c07a237b74442","observation_id":"42bcc413-7aba-417d-b587-4b64c1471ef6","resolution":{"observed_at":"2026-08-10T20:34:39.183354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.163198Z","title":"CLaM-TTS: Improving neural codec language model for zero-shot text-to-speech,","venue":null,"work_id":"0d1874a5-2773-460c-9632-75d820c7d316","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.391388Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:7dadaed94c8b3320fffa3c3c3de4078fd2ba98c2ce4e04c2d9e4ac4eed5fca4d","observation_id":"f970406f-d968-4ba3-832a-f79f3603c137","resolution":{"observed_at":"2026-08-10T20:34:39.168139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02702","last_updated":"2024-11-21T10:31:03Z","snapshot_observed_at":"2026-08-11T08:06:49.773907Z","submitted_at":"2024-04-03T13:00:08Z","title":"PSCodec: A Series of High-Fidelity Low-bitrate Neural Speech Codecs Leveraging Prompt Encoders","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02702","snapshot_observed_at":"2026-08-10T20:34:37.397522Z","title":"PromptCodec: High-fidelity neural speech codec using disentangled representation learning based adaptive feature- aware prompt encoders,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.397522Z"},"links":{"cited_paper":"/paper/2404.02702","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:a46bd76fb3c7b404151a7733dd974cdfbeb8bf7e75fb4f797e802b3390893150","observation_id":"28009564-8db2-4778-b015-a3d4b6ac2a15","resolution":{"observed_at":"2026-08-10T20:34:37.397522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.147709Z","title":"Addressing index collapse of large-codebook speech tokenizer with dual-decoding product-quantized variational auto-encoder,","venue":null,"work_id":"8c4b3d70-7fb7-4531-8ad7-8a226b323e31","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.404156Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:37695547c9812c43e1cf281459be492fbe8b1ff8d8cb897075746eed4082d506","observation_id":"cdae8398-6c77-4d9d-b1b6-c3170d8293d5","resolution":{"observed_at":"2026-08-10T20:34:39.153080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.132007Z","title":"SuperCodec: A neural speech codec with selective back-projection network,","venue":null,"work_id":"47718f41-d7c4-43f2-bf5e-16c58bbc02bf","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.410991Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:84c21e40849b72e9afc91814e205e142129b4fabe0c82bc9b6767ef3f92b7428","observation_id":"9e9a5363-8150-4175-8e5f-32c4a3aba726","resolution":{"observed_at":"2026-08-10T20:34:39.137126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04051","last_updated":"2024-07-11T02:08:35Z","snapshot_observed_at":"2026-08-06T00:42:36.144034Z","submitted_at":"2024-07-04T16:49:02Z","title":"FunAudioLLM: Voice Understanding and Generation Foundation Models for Natural Interaction Between Humans and LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04051","snapshot_observed_at":"2026-08-10T20:34:37.415881Z","title":"Funaudiollm: V oice understanding and generation foundation models for natural interaction between humans and llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.415881Z"},"links":{"cited_paper":"/paper/2407.04051","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:92e13c3b5f3a87c66564d4b9cc6a191b29cbbc8bb5e02cd4f04aaeaae654b674","observation_id":"09a83c99-b280-497d-b042-c39cc588b82b","resolution":{"observed_at":"2026-08-10T20:34:37.415881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.116249Z","title":"Learning source disentanglement in neural audio codec,","venue":null,"work_id":"c2678936-aab0-4ee3-ac39-f3d8f9518342","year":2025},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.423133Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:8ae5a4b34c306e1ee8c89245ebf02fb25834e8c5c1cb40f5dae2b9bc2b63a980","observation_id":"c94462db-769c-4682-a994-465c21ad9bf7","resolution":{"observed_at":"2026-08-10T20:34:39.121249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12121","last_updated":"2024-12-27T07:42:35Z","snapshot_observed_at":"2026-07-06T19:17:41.512834Z","submitted_at":"2024-09-18T16:45:09Z","title":"WMCodec: End-to-End Neural Speech Codec with Deep Watermarking for Authenticity Verification","version":3},"cited_work":{"arxiv_id":"2409.12121","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.12121","snapshot_observed_at":"2026-08-10T20:34:38.176179Z","title":"WMCodec: End-to-End Neural Speech Codec with Deep Watermarking for Authenticity Verification","venue":"cs.SD","work_id":"933846b0-9ce7-48d6-add2-bca36409bd49","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.429378Z"},"links":{"cited_paper":"/paper/2409.12121","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:90e7472f5429446031fc572348e885ec6a327164d3409e6ca80c4adaf4726c1c","observation_id":"96830127-f52a-4553-aff0-1f9d9ee18e1b","resolution":{"observed_at":"2026-08-10T20:34:38.184949Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12117","last_updated":"2024-09-18T16:39:10Z","snapshot_observed_at":"2026-08-05T08:42:30.244745Z","submitted_at":"2024-09-18T16:39:10Z","title":"Low Frame-rate Speech Codec: a Codec Designed for Fast High-quality Speech LLM Training and Inference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12117","snapshot_observed_at":"2026-08-10T20:34:37.434724Z","title":"Low Frame-rate Speech Codec: a codec designed for fast high-quality speech llm training and inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.434724Z"},"links":{"cited_paper":"/paper/2409.12117","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:2d668e5214f623ec312488ad13b43222a475080cec2e2c1eb129bf8f6aaee836","observation_id":"34a4f62f-6a52-4b1e-86c5-63dc3326e6ba","resolution":{"observed_at":"2026-08-10T20:34:37.434724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12717","last_updated":"2024-09-19T12:41:30Z","snapshot_observed_at":"2026-07-06T19:18:08.790424Z","submitted_at":"2024-09-19T12:41:30Z","title":"NDVQ: Robust Neural Audio Codec with Normal Distribution-Based Vector Quantization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12717","snapshot_observed_at":"2026-08-10T20:34:37.439760Z","title":"NDVQ: Robust neural audio codec with normal distribution-based vector quantization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.439760Z"},"links":{"cited_paper":"/paper/2409.12717","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:f4de5d987e875a1cb55dcd0d3145b6c7d1783dbf9171c44b6d0450a7325c820d","observation_id":"550f3fac-b037-4494-8a86-a10e1092c220","resolution":{"observed_at":"2026-08-10T20:34:37.439760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.100128Z","title":"Speechtokenizer: Unified speech tokenizer for speech language models,","venue":null,"work_id":"a094ad78-290a-4b70-877d-a906a0d272e3","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.445106Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:4682cd587111ff203481ac21bc596bffd80eeabf9d8c315ae94ca10197e8ada1","observation_id":"0b705f90-a235-4961-8836-b998477614e1","resolution":{"observed_at":"2026-08-10T20:34:39.105419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.086898Z","title":"High- fidelity audio compression with improved rvqgan,","venue":null,"work_id":"24c8b7a4-04d5-45a6-a0f6-e01960e9278d","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.451271Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:702024c1beb4c2e75018f7b2ab66e5ebebd742615f14e860dd945ebe5df5b50d","observation_id":"a3a61202-e085-48a3-8c63-8c31ccda093e","resolution":{"observed_at":"2026-08-10T20:34:39.091147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.02765","last_updated":"2023-05-07T09:22:04Z","snapshot_observed_at":"2026-08-06T10:45:04.364170Z","submitted_at":"2023-05-04T12:11:13Z","title":"HiFi-Codec: Group-residual Vector quantization for High Fidelity Audio Codec","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.02765","snapshot_observed_at":"2026-08-10T20:34:37.457209Z","title":"Hifi-codec: Group-residual vector quantization for high fidelity audio codec,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.457209Z"},"links":{"cited_paper":"/paper/2305.02765","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:5afa16402421448694629f7c4099c81fb707982b7f912b486a48951dcedf9ecc","observation_id":"9650cd48-a466-44e4-addc-b2441bc9eb6e","resolution":{"observed_at":"2026-08-10T20:34:37.457209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.073152Z","title":"High fidelity neural audio compression,","venue":null,"work_id":"77a17685-336d-403e-af62-562e6327b5ed","year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.462776Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:ac8faf4c767224120b3ec69f6f8018cd99f018266557d559e10e4e0eaf9518cb","observation_id":"ba3c08b0-1b7d-43a6-83bb-2340df06ed83","resolution":{"observed_at":"2026-08-10T20:34:39.077463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.057202Z","title":"SemantiCodec: An ultra low bitrate semantic audio codec for general sound,","venue":null,"work_id":"e47f797b-0195-451f-bb6a-e27632829d85","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.468487Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:b6317db2f81fc52b810e0e482bea8302f5eb01763635e6a188ef5b06bb3c747a","observation_id":"0508a9f5-cba6-4964-80a4-4c77a1993fa0","resolution":{"observed_at":"2026-08-10T20:34:39.062660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.041201Z","title":"Funcodec: A fundamental, reproducible and integrable open-source toolkit for neural speech codec,","venue":null,"work_id":"f89276f0-3d70-460f-a8ce-d157a157127d","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.473766Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:986f7d750f5b87cb96c9f3788cf0c8fe4e7b2ae171ebc8e57ba92add0592ff2e","observation_id":"a4fbd2c4-b210-41c4-a4b5-0133288402a2","resolution":{"observed_at":"2026-08-10T20:34:39.046288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.025150Z","title":"Naturalspeech 3: Zero-shot speech synthesis with factorized codec and diffusion models,","venue":null,"work_id":"ff237e00-742b-4df1-8682-277f5be06e91","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.478627Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:42d54e9665c01f7b06e2ee721ffd56ca8e79fed6d3671dbead55942d2870463a","observation_id":"fae1ad89-ddbc-4fce-a02c-a45fb1be2582","resolution":{"observed_at":"2026-08-10T20:34:39.030589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:39.008949Z","title":"UniAudio 1.5: Large language model-driven audio codec is a few-shot audio task learner,","venue":null,"work_id":"035331bb-7f4e-4c02-bf62-0b5de218472c","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.483384Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:2fab2398b647c8078a51cac0914ed7c848c3c867c24644fa6eb1d43a3cb0a4ab","observation_id":"896b4d0a-48bc-4097-8b1c-a7653194c656","resolution":{"observed_at":"2026-08-10T20:34:39.013741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.992721Z","title":"WavTok- enizer: an efficient acoustic discrete codec tokenizer for audio language modeling,","venue":null,"work_id":"ce4b6adf-c0d1-4869-acf6-6e5238ac4b6b","year":2025},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.488006Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:c47e339977f3b7020c8021485809dbff3def6936d7bee9b2d89cdbac9b43285f","observation_id":"2376bb00-6868-4915-af3b-eda46595dada","resolution":{"observed_at":"2026-08-10T20:34:38.998095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12208","last_updated":"2025-06-04T05:50:15Z","snapshot_observed_at":"2026-08-10T22:55:55.008318Z","submitted_at":"2024-02-19T15:12:12Z","title":"Language-Codec: Bridging Discrete Codec Representations and Speech Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12208","snapshot_observed_at":"2026-08-10T20:34:37.493331Z","title":"Language- codec: Reducing the gaps between discrete codec representation and speech language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.493331Z"},"links":{"cited_paper":"/paper/2402.12208","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:f780f254a59c101462f77e078e2ff6629a67afe67362f540b921b66f526d216b","observation_id":"541a6e41-8c4f-4da3-a894-985e54bdf85c","resolution":{"observed_at":"2026-08-10T20:34:37.493331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.975406Z","title":"Hilcodec: High-fidelity and lightweight neural audio codec,","venue":null,"work_id":"4ff1fd4c-eef2-4e83-b925-5169e8f70d96","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.498215Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:e563e4d503a171566dab16f6c1509f3d56ce3a3198fd259c271bffc06de3ac27","observation_id":"51dff18d-d7fb-4f86-a35f-b3ea28bfb64b","resolution":{"observed_at":"2026-08-10T20:34:38.981062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.960411Z","title":"V ocos: Closing the gap between time-domain and fourier- based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":"921e949e-630b-48e6-b909-fe1f7b96e514","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.502651Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:2312a4bad227fc8688b5ce3018aa710eab57861078242056577139026797ef29","observation_id":"9a8e3121-90ab-45fa-b8e2-a88c3b43995b","resolution":{"observed_at":"2026-08-10T20:34:38.965506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.945193Z","title":"Socodec: A semantic-ordered multi-stream speech codec for efficient language model based text-to-speech synthesis,","venue":null,"work_id":"d1e4a115-8c6c-4103-82ba-1aa52a7328ac","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.506651Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:7adb421892762a9ced55a2ba4616ee4fb76c5fab564649479a3971849e8cefd8","observation_id":"b5d667e9-1555-4844-aebb-bd40c5c73d31","resolution":{"observed_at":"2026-08-10T20:34:38.950547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.929199Z","title":"SNAC: Multi-scale neural audio codec,","venue":null,"work_id":"860a8b64-ef4d-4fc2-9353-8e3bd0e1aeeb","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.510805Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:14e8d87a5092c555886597b32cd0a5a4c32d82dc2ea95143fbdef758e4c0eb4a","observation_id":"8a0f6660-022e-49bb-80dd-4ef59390476f","resolution":{"observed_at":"2026-08-10T20:34:38.934307Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17175","last_updated":"2024-11-27T11:47:45Z","snapshot_observed_at":"2026-08-10T14:08:31.424118Z","submitted_at":"2024-08-30T10:24:07Z","title":"Codec Does Matter: Exploring the Semantic Shortcoming of Codec for Audio Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17175","snapshot_observed_at":"2026-08-10T20:34:37.516552Z","title":"Codec Does Matter: Exploring the semantic shortcoming of codec for audio language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.516552Z"},"links":{"cited_paper":"/paper/2408.17175","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:884a2bbfc08d4a2b1a7dc760e2bf00b4d2c1ce5375dcadd3f0de2ff1af1b9dc5","observation_id":"1680dde7-adbd-41b8-8903-a5ab7260ce15","resolution":{"observed_at":"2026-08-10T20:34:37.516552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.913959Z","title":"Audiodec: An open-source streaming high-fidelity neural audio codec,","venue":null,"work_id":"ae9c31ec-61cd-4651-983b-23c7f03518f2","year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.521670Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:da0769e03583acae2c271726f993c79462dac433d7cc69bb6814d60fcec659f8","observation_id":"b2e9d9bf-f55d-47f4-94c2-22cb58185f04","resolution":{"observed_at":"2026-08-10T20:34:38.919182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.05377","last_updated":"2024-09-09T07:18:07Z","snapshot_observed_at":"2026-08-03T20:35:31.539481Z","submitted_at":"2024-09-09T07:18:07Z","title":"BigCodec: Pushing the Limits of Low-Bitrate Neural Speech Codec","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.05377","snapshot_observed_at":"2026-08-10T20:34:37.527498Z","title":"BigCodec: Push- ing the limits of low-bitrate neural speech codec,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.527498Z"},"links":{"cited_paper":"/paper/2409.05377","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:6c93248eeca137b7330d8e2809220ca18e5beb8874b9e7f5038d0b91aaeff181","observation_id":"183c0cf4-00e2-4fe6-98de-43313629bdfd","resolution":{"observed_at":"2026-08-10T20:34:37.527498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.898374Z","title":"Fewer- token neural speech codec with time-invariant codes,","venue":null,"work_id":"52d73260-bf48-4e67-86b6-3c11157748c6","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.532826Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:c09c84b3c80b3b4343769eac4e11b32fd190865feb07368c4059b3137014d73c","observation_id":"256c992d-70dc-4e45-92bb-7ad457f887b9","resolution":{"observed_at":"2026-08-10T20:34:38.903493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-10T20:34:37.537842Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.537842Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:c3c284ba70c91abb5c5e56294c4838cdc66cd56f8bf6729c0bdecaf9cb809581","observation_id":"1633e10f-4060-45be-a244-40d2df1dfe6e","resolution":{"observed_at":"2026-08-10T20:34:37.537842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05298","last_updated":"2025-06-04T16:25:54Z","snapshot_observed_at":"2026-07-06T18:27:24.526210Z","submitted_at":"2024-06-07T23:47:51Z","title":"Spectral Codecs: Improving Non-Autoregressive Speech Synthesis with Spectrogram-Based Audio Codecs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05298","snapshot_observed_at":"2026-08-10T20:34:37.542880Z","title":"Spectral Codecs: Spectrogram-based audio codecs for high quality speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.542880Z"},"links":{"cited_paper":"/paper/2406.05298","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:8726c5bc094b17a3db018000100fcea956a5cc8be4766943a8aefd4781f96c87","observation_id":"390aab34-b3dd-4d42-86e3-06b2f8f58bc8","resolution":{"observed_at":"2026-08-10T20:34:37.542880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.882328Z","title":"Simple- speech: Towards simple and efficient text-to-speech with scalar latent transformer diffusion models,","venue":null,"work_id":"30c57b94-6589-41b1-9e8d-46edfc5b8a2d","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.547968Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:72eec10729d7a59ed328bc0193198e89a7be303030a6f78643481d3019cc9880","observation_id":"02e80d2c-bb89-48ff-8649-f0f25fc964ce","resolution":{"observed_at":"2026-08-10T20:34:38.887157Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.867178Z","title":"Single- codec: Single-codebook speech codec towards high-performance speech generation,","venue":null,"work_id":"0079f15a-9d3d-4c40-87f6-d9329c3c1bf9","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.552576Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:fa46e57d9c176a0f91eb8625f20dc8cf9a72c5c4a1250eb581daeb090af80c36","observation_id":"f19deefa-eb69-4ded-8187-0737caa7d89a","resolution":{"observed_at":"2026-08-10T20:34:38.872502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18803","last_updated":"2025-04-27T16:37:28Z","snapshot_observed_at":"2026-08-06T16:58:49.935180Z","submitted_at":"2024-11-27T23:07:52Z","title":"TS3-Codec: Transformer-Based Simple Streaming Single Codec","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.18803","snapshot_observed_at":"2026-08-10T20:34:37.557233Z","title":"Ts3-codec: Transformer- based simple streaming single codec,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.557233Z"},"links":{"cited_paper":"/paper/2411.18803","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:8c6461af21796e6cbad9d54cf3b6ef7f588acbcc9bb4cbd9175f2f2f3fd093d2","observation_id":"4c2ab74e-6c20-497b-b9d0-8e18c6213b42","resolution":{"observed_at":"2026-08-10T20:34:37.557233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.852555Z","title":"Codec-SUPERB: An in-depth analysis of sound codec models,","venue":null,"work_id":"11ec71a7-1f22-4655-8ad6-8ed7df9415f2","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.562186Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:3dd4155495928b0328e20d4f093a99d4bb7e9a6c8ffe7f39bb30595ad254e4e5","observation_id":"a7e73a26-6cd8-448b-817f-49acd25f709d","resolution":{"observed_at":"2026-08-10T20:34:38.857203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14294","last_updated":"2026-04-21T15:27:17Z","snapshot_observed_at":"2026-07-30T17:14:02.349402Z","submitted_at":"2024-06-20T13:23:27Z","title":"DASB - Discrete Audio and Speech Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14294","snapshot_observed_at":"2026-08-10T20:34:37.566999Z","title":"DASB–discrete audio and speech benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.566999Z"},"links":{"cited_paper":"/paper/2406.14294","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:5a6039879f4c674217f05e06fd959cbddb43fbd2dcd17490aac273c95f0dd162","observation_id":"ed7f0b3b-f296-44bc-8303-6ae8ec9b894d","resolution":{"observed_at":"2026-08-10T20:34:37.566999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.836834Z","title":"Codec- SUPERB@ SLT 2024: A lightweight benchmark for neural audio codec models,","venue":null,"work_id":"888e30ce-efb5-44eb-b6d4-09ba3c7db9fc","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.571475Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:ecdac761f7e6db79679e3c252d53e4b122a814c935e763f146595b6eb20414c7","observation_id":"401dd573-11b5-49e2-b85e-a1c78912a6dd","resolution":{"observed_at":"2026-08-10T20:34:38.842201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.821087Z","title":"ESPnet-Codec: Comprehensive training and evaluation of neural codecs for audio, music, and speech,","venue":null,"work_id":"0b3d299b-baff-46e0-8626-2d00d4df16e5","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.575669Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:c519832d8f79cb21db6aaafefba2a0515f0fa0a32499057aa3616ba3684c19a4","observation_id":"c7b18340-4cb9-4c36-82a1-10f64ddfc695","resolution":{"observed_at":"2026-08-10T20:34:38.826343Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.17667","last_updated":"2025-03-27T03:50:10Z","snapshot_observed_at":"2026-08-11T15:07:28.270038Z","submitted_at":"2024-12-23T15:53:21Z","title":"VERSA: A Versatile Evaluation Toolkit for Speech, Audio, and Music","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.17667","snapshot_observed_at":"2026-08-10T20:34:37.579626Z","title":"Versa: A versatile evaluation toolkit for speech, audio, and music,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.579626Z"},"links":{"cited_paper":"/paper/2412.17667","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:c670e42f602c0d342adf89ec206dfcca93aba1b873cfad0a3706f1ec8dd64a1d","observation_id":"0bd20adb-da0f-4bc5-927a-28128590099a","resolution":{"observed_at":"2026-08-10T20:34:37.579626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13236","last_updated":"2024-02-20T18:50:25Z","snapshot_observed_at":"2026-08-06T03:13:16.529776Z","submitted_at":"2024-02-20T18:50:25Z","title":"Towards audio language modeling -- an overview","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13236","snapshot_observed_at":"2026-08-10T20:34:37.584349Z","title":"Towards audio language modeling-an overview,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.584349Z"},"links":{"cited_paper":"/paper/2402.13236","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:9288da331686098afb39fd2419d59aa34b572774b70b6bc409a7f19bfd8556f1","observation_id":"a39d7d30-0c0a-4030-ba74-069b38e6b5d2","resolution":{"observed_at":"2026-08-10T20:34:37.584349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.805029Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":"04b77ba7-372e-4aad-ab35-cab853064787","year":2025},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.589062Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:9e56ae30727409ca8c977c25f1a3971b97a29a9048e843e31e4c4f512f085bbf","observation_id":"4f552e06-2af7-4e8c-8dd1-ce4be3c5bf4f","resolution":{"observed_at":"2026-08-10T20:34:38.810096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03926","last_updated":"2023-03-07T14:31:55Z","snapshot_observed_at":"2026-08-06T04:55:08.186019Z","submitted_at":"2023-03-07T14:31:55Z","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03926","snapshot_observed_at":"2026-08-10T20:34:37.593743Z","title":"Speak foreign languages with your own voice: Cross-lingual neural codec language modeling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.593743Z"},"links":{"cited_paper":"/paper/2303.03926","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:6033f1cd0b39080974d3c67cd324d835244e5c96653865a281a2da5956dba43e","observation_id":"701c5524-63f6-43d6-995f-2876537a8b5c","resolution":{"observed_at":"2026-08-10T20:34:37.593743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.788780Z","title":"Viola: Conditional language models for speech recognition, synthesis, and 15 translation,","venue":null,"work_id":"3f9aa87f-a5b5-4878-bf76-aa161360f243","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.598434Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:8c57a6d7be1979718fa1576edfca50ca8c4d21bd075ca9f17096ac555a0be0d1","observation_id":"b7703f36-0189-4c33-abdc-49e21d7e2d8e","resolution":{"observed_at":"2026-08-10T20:34:38.794448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.772530Z","title":"UniAudio: Towards universal audio generation with large language models,","venue":null,"work_id":"2fb0dd67-cd59-4635-8a03-37147ca0fecd","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.602496Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:47e3e2f966023593ee4d7c08a9ff036c403aa69fe08c83725a06364943413ab5","observation_id":"437b6ff5-f581-41b2-a29b-293cb07d0db2","resolution":{"observed_at":"2026-08-10T20:34:38.777638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.04673","last_updated":"2024-07-03T02:38:03Z","snapshot_observed_at":"2026-08-10T13:50:38.422341Z","submitted_at":"2023-10-07T03:17:59Z","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.04673","snapshot_observed_at":"2026-08-10T20:34:37.606605Z","title":"Lauragpt: Listen, attend, understand, and regenerate audio with gpt,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.606605Z"},"links":{"cited_paper":"/paper/2310.04673","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:e2d3c54f6bd68d40d2367f2fe02e79c65d61bafb310d7c4a1dbe0b31b0a976f2","observation_id":"918f8e07-742d-4f2e-902c-5b6441582155","resolution":{"observed_at":"2026-08-10T20:34:37.606605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.756334Z","title":"Speechx: Neural codec language model as a versatile speech transformer,","venue":null,"work_id":"8d937faf-e44d-4d5d-a024-0f5e80afaa0a","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.610777Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:99d060266e8720baf59c529977fd0e98a2967b13520f622949810934f21de8d9","observation_id":"d0e1ce39-0893-4d8c-ab3f-25ad117283d1","resolution":{"observed_at":"2026-08-10T20:34:38.761485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.742196Z","title":"Does audio deepfake detection generalize?","venue":null,"work_id":"d4ebf0c4-3554-4bc3-9fd6-6079546e4d6e","year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.615284Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:e580d7157db930d713fd70474954595e95b9c4f7cfdddbc9e52ac3f6197951f1","observation_id":"e94ed9b6-275d-4e02-9eca-1287469828a5","resolution":{"observed_at":"2026-08-10T20:34:38.746935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.727185Z","title":"Codecfake: An initial dataset for detecting llm-based deepfake audio,","venue":null,"work_id":"7e4b6f80-4e8c-4dbb-8a1f-6eef1d01e889","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.619932Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:7be491ac8ed21e905160f61897a9f6700e32eb629df10991d07c78827e88df2f","observation_id":"22d664d6-636c-4351-94dc-f826f286f64a","resolution":{"observed_at":"2026-08-10T20:34:38.732086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.711632Z","title":"The codecfake dataset and countermeasures for the universally detection of deepfake audio,","venue":null,"work_id":"38a6a0e0-a67e-4a10-a387-1452bbe19d49","year":2025},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.624840Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:0189f5488e00b7802cfd7911f0320ac62af0952a09819ad933d2772aa90ca00f","observation_id":"9cbad0bd-5f5b-4171-936f-30f3ce685eec","resolution":{"observed_at":"2026-08-10T20:34:38.717172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.695883Z","title":"ASVspoof 2021: Towards spoofed and deepfake speech detection in the wild,","venue":null,"work_id":"b2891ffd-5109-41bb-ab56-f176d5235f3f","year":2021},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.629274Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:2badad97dee3bb4e31b37e2e89c583bb7f9d437079e79a1d02d51bfe1b11eab3","observation_id":"fc7ce245-24dc-4a8d-a38f-ce5f39ed21c8","resolution":{"observed_at":"2026-08-10T20:34:38.700956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.678509Z","title":"CFAD: A chinese dataset for fake audio detection,","venue":null,"work_id":"7c832115-75f6-4fef-85a2-4aeacae8b91e","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.634085Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:8f7dd45334e2e4355f1621882d2c1a9b79b29af0bccd565198f52da9d85a86e7","observation_id":"77ac6406-731e-4a62-8108-fae3efa9f28f","resolution":{"observed_at":"2026-08-10T20:34:38.683621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.662344Z","title":"Mlaad: The multi-language audio anti-spoofing dataset,","venue":null,"work_id":"c5c59993-6aad-4ea2-a1b8-eba30c23665d","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.638453Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:1cd1fe134a73e3c5753077c919ac94bd16ec5fa105e26e9612d5fec94333f804","observation_id":"c44dd764-d225-4b79-93c9-3becf010bdfc","resolution":{"observed_at":"2026-08-10T20:34:38.667552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.646448Z","title":"ASVspoof 5: Crowdsourced speech data, deepfakes, and adversarial attacks at scale,","venue":null,"work_id":"a25b8b89-b930-4806-9adf-7c0b62964045","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.643109Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:15c37f71f9735bcd5e02c1d566b31a6d99f436f43250892db69ae5ee37eae254","observation_id":"c687d3c7-723a-4d46-b714-597600ccbdd7","resolution":{"observed_at":"2026-08-10T20:34:38.651564Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.630251Z","title":"DFADD: The diffusion and flow-matching based audio deepfake dataset,","venue":null,"work_id":"a35a0318-0920-4d4d-bb84-fae016cecd2c","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.647327Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:cbe6cbf5b5cae39ff89c24153c40f55f6875ac5003a1bff2102c6a00c1f2b582","observation_id":"8bef9b30-a389-471f-b8a8-82818d33aa92","resolution":{"observed_at":"2026-08-10T20:34:38.635231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07333","last_updated":"2024-01-14T17:43:55Z","snapshot_observed_at":"2026-08-11T13:56:23.370106Z","submitted_at":"2024-01-14T17:43:55Z","title":"ELLA-V: Stable Neural Codec Language Modeling with Alignment-guided Sequence Reordering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07333","snapshot_observed_at":"2026-08-10T20:34:37.652047Z","title":"Ella-v: Stable neural codec language modeling with alignment-guided sequence reordering,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.652047Z"},"links":{"cited_paper":"/paper/2401.07333","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:128cad399d62307af1be2cbe74b0df9221b143878ed11f25f1b3bb8b6bdc6e9c","observation_id":"13d7ca2a-2279-434b-8f66-9cf72fd94b38","resolution":{"observed_at":"2026-08-10T20:34:37.652047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.611230Z","title":"Generative pre-trained speech language model with efficient hierarchical transformer,","venue":null,"work_id":"32c08f54-eaf3-4445-80c1-fc65c13beb2a","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.656856Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:81630fd03f975c327d797b640de50f08a8270d22f08a644b5f7e655898d28756","observation_id":"cf671fea-e5b2-47cb-bd9e-59277f7c413a","resolution":{"observed_at":"2026-08-10T20:34:38.616905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.592547Z","title":"TacoLM: Gated attention equipped codec language model are efficient zero-shot text to speech synthesizers,","venue":null,"work_id":"249ff1c9-73fc-4100-a5b9-17819114f4ef","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.661585Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:22eb0e46e873ac74ef00ab711e63b63afb3284863d5ef9484159273e83cf31e8","observation_id":"140d876c-7231-4491-8c7e-9f51ba1da624","resolution":{"observed_at":"2026-08-10T20:34:38.597946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03204","last_updated":"2024-05-19T21:34:28Z","snapshot_observed_at":"2026-08-10T13:53:18.848991Z","submitted_at":"2024-04-04T05:15:07Z","title":"RALL-E: Robust Codec Language Modeling with Chain-of-Thought Prompting for Text-to-Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03204","snapshot_observed_at":"2026-08-10T20:34:37.666439Z","title":"Rall-e: Robust codec language modeling with chain-of-thought prompting for text-to- speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.666439Z"},"links":{"cited_paper":"/paper/2404.03204","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:5f0820b5e13f3fcfa2b95aaab32d407bd42ac6b71433058de80ee063241db586","observation_id":"8e95e3fd-ad05-45c2-94f6-040fe614f416","resolution":{"observed_at":"2026-08-10T20:34:37.666439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13893","last_updated":"2024-08-28T07:16:37Z","snapshot_observed_at":"2026-07-31T16:15:52.677226Z","submitted_at":"2024-08-25T17:07:39Z","title":"SimpleSpeech 2: Towards Simple and Efficient Text-to-Speech with Flow-based Scalar Latent Transformer Diffusion Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13893","snapshot_observed_at":"2026-08-10T20:34:37.671286Z","title":"Simplespeech 2: Towards simple and efficient text-to-speech with flow-based scalar latent transformer diffusion models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.671286Z"},"links":{"cited_paper":"/paper/2408.13893","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:f4ef83d05f929bf01e92534a62a25e8ed812dfeca43a6c391d94acc75217c938","observation_id":"5ae7b87e-a029-49ba-8ef0-20ddb5d6da4b","resolution":{"observed_at":"2026-08-10T20:34:37.671286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.576778Z","title":"MaskGCT: Zero-shot text-to-speech with masked generative codec transformer,","venue":null,"work_id":"33b28365-1a5d-4a4f-96de-56da21a99010","year":2025},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.676241Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:31f5f0ce997ddf3f1d5771ab2d6a72422b034f7067e8a0d6bfed51c759414308","observation_id":"3aa7d816-b625-49cb-8eec-60aa3460c3a3","resolution":{"observed_at":"2026-08-10T20:34:38.581915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.561114Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers,","venue":null,"work_id":"d52c6d3c-b923-4af2-8540-f2e295ba72df","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.681286Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:50abf8922b5b16048c850f94365f2018abdb533d10eb74806ffd25c518d59e1c","observation_id":"6264bae6-ad19-4f3f-98aa-5abad83d17af","resolution":{"observed_at":"2026-08-10T20:34:38.566285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.544823Z","title":"E2 TTS: Embarrassingly easy fully non-autoregressive zero-shot TTS,","venue":null,"work_id":"64fb141e-4823-4b89-a232-19d97b9a4d72","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.685784Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:ff32684d9009c23434f8f4ba955bd8e6918c87333b7415d014a3ccc0547bb87d","observation_id":"400c7445-16e3-4577-b15d-548a9efbc3ef","resolution":{"observed_at":"2026-08-10T20:34:38.549775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.528276Z","title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":"b0c6bf77-66e5-458d-b058-9c22b6354b3f","year":2020},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.690106Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:889a88730df39c2e73d690a192a8c56aa48978d7e22f13b28f77777bc3bcdcf5","observation_id":"0b612adc-e7df-4a1d-a31f-74d6edf3197b","resolution":{"observed_at":"2026-08-10T20:34:38.533953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.510341Z","title":"ASVspoof 2019: A large-scale public database of synthesized, converted and replayed speech,","venue":null,"work_id":"830a7841-10f7-47bc-8a15-2a43a6d48b39","year":2019},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.694435Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:633392394a606c5a30ea31e1b816637c9e2951b3cf7f59b4e8c385138d987450","observation_id":"d2c0f1e9-e8cd-4c36-8b80-ec4773e1e375","resolution":{"observed_at":"2026-08-10T20:34:38.515734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.495821Z","title":"Cstr vctk corpus: English multi-speaker corpus for cstr voice cloning toolkit (version 0.92),","venue":null,"work_id":"8fb29eb6-d996-4313-9951-bd54242dac87","year":2019},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.698588Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:f7b23b8be0f564bd4665d1963c5d6b69d412cf36804d9cb4d258d915cde27fda","observation_id":"299eb0c8-fca9-4087-9e5d-3ae5489ec64b","resolution":{"observed_at":"2026-08-10T20:34:38.500452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.703052Z","title":"Ditar: Diffusion transformer autoregressive modeling for speech generation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.703052Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:0b95c1fea938c81b10731d602ec913178f099b8a50daef4e2b3df4efba1a5e3a","observation_id":"0c4161fd-e94a-48bc-b438-a44415f24417","resolution":{"observed_at":"2026-08-10T20:34:37.703052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.476032Z","title":"Masked autoencoders that listen,","venue":null,"work_id":"f54b98ae-17ca-4b9f-bb34-62c20b980372","year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.707325Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:7e37d160441da507a7fbe3f2d69fb87854a0185ebd4ec71e144d9a5f0b686b88","observation_id":"8efec6a4-b713-4801-b53d-5304c4cb17c8","resolution":{"observed_at":"2026-08-10T20:34:38.482256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.457239Z","title":"W2v-bert: Combining contrastive learning and masked language mod- eling for self-supervised speech pre-training,","venue":null,"work_id":"26257fa6-c8bc-49fb-b6af-39ea6e79fb07","year":2021},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.711639Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:22c398c44e5e5d3f4196ff2f11a482b7e1eadf3515adb9dd611a2f80206071c5","observation_id":"645da801-e000-4a74-99b8-814811b38291","resolution":{"observed_at":"2026-08-10T20:34:38.463241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.440488Z","title":"Vector-quantized image modeling with im- proved VQGAN,","venue":null,"work_id":"7259ad94-2809-4437-8718-24d765d892bc","year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.716603Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:1252f24a93949337482d0ee73916fbebd633660148ca7713efb404fed100c90e","observation_id":"49672bad-f15b-4419-988e-ddf22ba6e32f","resolution":{"observed_at":"2026-08-10T20:34:38.445869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-10T20:34:37.721324Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.721324Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:c64bf887961104aa373cae672011367a2e5493ae6dd164fc0ffe3f14a73af613","observation_id":"6ce102c0-b3fa-4330-bd8a-2d79f29640e7","resolution":{"observed_at":"2026-08-10T20:34:37.721324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.422355Z","title":"Amphion: An open-source audio, music and speech generation toolkit,","venue":null,"work_id":"e4cdc3b9-3df0-4ff2-9435-92346726916d","year":2024},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.725679Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:1e322e2c2587886388c1fbb409ae40491e6bc0ec899475c289316b4f3c2f2a23","observation_id":"c89a8ee8-710a-4ddb-a203-b1dcd22b0360","resolution":{"observed_at":"2026-08-10T20:34:38.428054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.402192Z","title":"Utmos: Utokyo-sarulab system for voicemos challenge 2022,","venue":null,"work_id":"b43a9592-5920-4e04-adec-a4d92cae706b","year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.730221Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:3127c9c949e5bac85a2d9ab209684ac1f9960156034c93d39a0beafe0467b83a","observation_id":"add57893-beff-484d-a3dd-c9d06c9269b8","resolution":{"observed_at":"2026-08-10T20:34:38.407681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.385049Z","title":"The voicemos challenge 2022,","venue":null,"work_id":"a5738217-d5d1-4a80-a942-c9b2c94ab0ee","year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.734546Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:4699e17470d318b1766118c13b2408bed0e5747988a32ba771ff27c24d05af5d","observation_id":"426cd785-5bbb-4674-9999-258891ab46fd","resolution":{"observed_at":"2026-08-10T20:34:38.390254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:37.739712Z","title":"Automatic speaker verification spoofing and deepfake detection using wav2vec 2.0 and data augmentation,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.739712Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:5e66549494ed3ef67a69fcfb05985ae68be2567cbd7766b168634d3a106ab7f4","observation_id":"4a0b0bf2-e72a-4b59-bb54-076a56e8c289","resolution":{"observed_at":"2026-08-10T20:34:37.739712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.357534Z","title":"Rawboost: A raw data boosting and augmentation method applied to automatic speaker verification anti-spoofing,","venue":null,"work_id":"45319de5-15e6-4ee5-9b97-2df9153f422a","year":2022},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.744351Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:0e6d1b9b7811efcfcaa751ea5fc4ae3e34c5ca5a8a0824aac4aae6a280c0201b","observation_id":"9ba541a3-da80-4f93-bf5e-0c4490e2ba0e","resolution":{"observed_at":"2026-08-10T20:34:38.362784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T20:34:38.340925Z","title":"Spoofed training data for speech spoofing countermeasure can be efficiently created using neural vocoders,","venue":null,"work_id":"c2ef5b8d-b6bd-4d08-91cd-a84831699cbf","year":2023},"citing_paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech","version":3},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:37.749011Z"},"links":{"citing_paper":"/paper/2501.08238"},"observation_digest":"sha256:e34a8dfb0077184c19692e4d19142137799d3457cadfae02d20270e1bbd89973","observation_id":"ac48f70a-68fa-4f94-be82-b4487fc67109","resolution":{"observed_at":"2026-08-10T20:34:38.346188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.08238","last_updated":"2026-06-05T17:32:46Z","latest_version":3,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-10T20:27:08.151956Z","submitted_at":"2025-01-14T16:26:14Z","title":"CodecFake+: Codec-Based Resynthesized Data as a Proxy for Detecting CodecFake Speech"},"reference_resolution":{"displayed":97,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":44,"verified_exact":1,"verified_fuzzy":52},"total_outbound_references":97},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 97 of 97 outbound references and 10 inbound Pith citation observations for arXiv:2501.08238."}