{"as_of":"2026-08-13T04:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:58b8253e00347bc8eff3de8e53525c7d0fba0ca174dfc5d36415fd5ead6e6b2c","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T17:24:22.463020Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.09032/citation-record","integrity":"/paper/2412.09032/integrity","json":"/paper/2412.09032/citation-record.json","paper":"/paper/2412.09032"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:23.030039Z","title":"Spoofing countermeasures for the protection of automatic speaker recognition systems against attacks with artificial signals","venue":null,"work_id":"9715b77b-6add-4a9c-96c1-f5e4f76a8f6e","year":2012},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.318395Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:d084388c5f998dc97dad6bb5d594ce99a278bf37ab662b24eb7b6ac671b425e2","observation_id":"a6d995bb-b1f7-427a-9a27-5f807032cc6c","resolution":{"observed_at":"2026-08-11T17:24:23.033528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.00501","last_updated":"2019-06-30T23:58:28Z","snapshot_observed_at":"2026-07-06T08:03:56.215960Z","submitted_at":"2019-06-30T23:58:28Z","title":"Deep Residual Neural Networks for Audio Spoofing Detection","version":1},"cited_work":{"arxiv_id":"1907.00501","doi":null,"metadata_source":"pith","pith_arxiv_id":"1907.00501","snapshot_observed_at":"2026-08-11T17:24:22.748115Z","title":"Deep Residual Neural Networks for Audio Spoofing Detection","venue":"cs.LG","work_id":"5d212263-75eb-4291-abeb-5529aef1c53f","year":2019},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.322620Z"},"links":{"cited_paper":"/paper/1907.00501","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:db693a8f5546e232de8e6fd6df01eace3b4803860b492cd8a427d4863a5ce198","observation_id":"92220004-9ea6-48da-83b1-897812b038be","resolution":{"observed_at":"2026-08-11T17:24:22.752022Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:23.020572Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","venue":null,"work_id":"fffabf76-d2a2-4d39-986c-0dd752393838","year":2020},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.326606Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:b2ae8aadfb241040cde09e07a24643a63022ea4715e9775a6e2995115babbb80","observation_id":"7a288da7-442b-4e0b-9d43-f38b0707a1c7","resolution":{"observed_at":"2026-08-11T17:24:23.024042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:23.010437Z","title":"Waveform boundary detection for partially spoofed audio","venue":null,"work_id":"76bdc5da-3bde-42a0-a679-69431732a200","year":2023},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.331982Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:d023ea98b47ef7836656f70ad0356c8a40ed08caaf150b34fcf3ed4ed7073b5b","observation_id":"08bb22c2-d929-4f43-85da-2a8cb6e023b4","resolution":{"observed_at":"2026-08-11T17:24:23.014011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10281","last_updated":"2023-08-20T14:29:04Z","snapshot_observed_at":"2026-08-08T23:13:32.294893Z","submitted_at":"2023-08-20T14:29:04Z","title":"The DKU-DUKEECE System for the Manipulation Region Location Task of ADD 2023","version":1},"cited_work":{"arxiv_id":"2308.10281","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.10281","snapshot_observed_at":"2026-08-11T17:24:22.732646Z","title":"The DKU-DUKEECE System for the Manipulation Region Location Task of ADD 2023","venue":"eess.AS","work_id":"c498d916-11d9-4a05-bd02-8f46c7a4e171","year":2023},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.335949Z"},"links":{"cited_paper":"/paper/2308.10281","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:d793f289bd7e98e7625cb3a33f14c7abcac4c2ec7a767c45f80d28aa75cbf6a4","observation_id":"31ad5b6f-cd86-46a4-9bd3-685ed3011d42","resolution":{"observed_at":"2026-08-11T17:24:22.737955Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:23.000689Z","title":"Wavlm: Large-scale self-supervised pre-training for full stack speech processing","venue":null,"work_id":"3a1bc290-d54e-496f-966a-f6ab9c8155a3","year":2022},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.339515Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:946dc9318a2abde62085441e0d5d8d4cc2025ecce74c0de059165829ec729472","observation_id":"e83486a8-5db2-4d13-94de-cc3036e57305","resolution":{"observed_at":"2026-08-11T17:24:23.004306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02813","last_updated":"2021-11-04T12:26:34Z","snapshot_observed_at":"2026-07-06T12:05:25.336577Z","submitted_at":"2021-11-04T12:26:34Z","title":"WaveFake: A Data Set to Facilitate Audio Deepfake Detection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.02813","snapshot_observed_at":"2026-08-11T17:24:22.342603Z","title":"Wavefake: A data set to facilitate audio deepfake detection","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.342603Z"},"links":{"cited_paper":"/paper/2111.02813","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:52371f8abc0e9e742b4593374f90bac2a47e47adb5d41c21c5dd9b38f95e9d81","observation_id":"dce39ae0-8621-465d-a8b5-71cad0019bd0","resolution":{"observed_at":"2026-08-11T17:24:22.342603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.990443Z","title":"Aasist: Audio anti-spoofing using integrated spectro-temporal graph attention networks","venue":null,"work_id":"abc7efb9-fe32-4491-bb99-ca8fd1cbe2e6","year":2022},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.345768Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:b626d88436ca6cff90f801c69dcd5fae1db44328fee9e62cee33550097b599d7","observation_id":"0f949aa9-723a-46e2-b233-7c94fbf6cddc","resolution":{"observed_at":"2026-08-11T17:24:22.994462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.14732","last_updated":"2022-03-28T13:19:14Z","snapshot_observed_at":"2026-08-04T10:06:42.149198Z","submitted_at":"2022-03-28T13:19:14Z","title":"SASV 2022: The First Spoofing-Aware Speaker Verification Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.14732","snapshot_observed_at":"2026-08-11T17:24:22.348820Z","title":"Sasv 2022: The first spoofing-aware speaker verification challenge","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.348820Z"},"links":{"cited_paper":"/paper/2203.14732","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:e4a11541f77657b46e22b36be5e2b1541d4e8e33549708ff360619e5c92767ad","observation_id":"677ef5ad-e3f5-4987-88df-068476287c2d","resolution":{"observed_at":"2026-08-11T17:24:22.348820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.980299Z","title":"Application of the electrical network frequency (enf) criterion: A case of a digital recording","venue":null,"work_id":"6588efb4-a95c-49a9-9695-3a34cf7e9f95","year":2005},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.352425Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:572bbdfa317077e2cf795b0e396746be06aa89157479ee1c997a235200b92bdb","observation_id":"4d921a96-c5b4-44d5-a66f-04cc061a0835","resolution":{"observed_at":"2026-08-11T17:24:22.984089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.969304Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","venue":null,"work_id":"aee7245a-8a0b-4b70-bb01-9db3e08eb7e9","year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.355852Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:eaca05f57b7106d97b50f32ca27d656271599883449c982e08ca7411b860a0df","observation_id":"2bc74128-ca60-4c2a-a669-98d15a4dae93","resolution":{"observed_at":"2026-08-11T17:24:22.973032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.05576","last_updated":"2019-04-11T08:37:43Z","snapshot_observed_at":"2026-08-12T02:23:44.222099Z","submitted_at":"2019-04-11T08:37:43Z","title":"STC Antispoofing Systems for the ASVspoof2019 Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.05576","snapshot_observed_at":"2026-08-11T17:24:22.359356Z","title":"Stc antispoofing systems for the asvspoof2019 challenge","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.359356Z"},"links":{"cited_paper":"/paper/1904.05576","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:42fcab07221d4c6f4f5fdfc5f04536d601648ce7b07c0042bb67730541f41235","observation_id":"856bb995-a342-4b06-b09d-253d83f6e8f3","resolution":{"observed_at":"2026-08-11T17:24:22.359356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.958780Z","title":"Multi-grained backend fusion for manipulation region location of partially fake audio","venue":null,"work_id":"cde50665-c929-413a-b46e-a901829ec25e","year":2023},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.363171Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:5474e8a9c964d3f9be7f2703b325ee937116d9cef7ab99288613989014bc5a72","observation_id":"4501229f-e201-4ef7-a9ce-834c6dceb045","resolution":{"observed_at":"2026-08-11T17:24:22.962699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.948623Z","title":"Convolutional recurrent neural network and multitask learning for manipulation region location","venue":null,"work_id":"fcb98b7c-3507-44de-8ddd-118c7366c137","year":2023},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.366490Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:e6b458388a18655f9dec119af7fd3cc7cf24697e47db03595b6196fb1715839c","observation_id":"267c147b-945d-44a8-a90b-86b30e79abda","resolution":{"observed_at":"2026-08-11T17:24:22.952333Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.938121Z","title":"Single shot temporal action detection","venue":null,"work_id":"7a490bae-30af-4e94-a2ee-e8fa551d27f8","year":2017},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.369678Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:ad17a7c64a2b70531f4e36aa20c142820314d29bd28e12a29a2e361a76d1f284","observation_id":"c09540c1-c756-42f6-b11b-ba6f089954f1","resolution":{"observed_at":"2026-08-11T17:24:22.942184Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.927574Z","title":"Gaussian temporal awareness networks for action localization","venue":null,"work_id":"af7615bd-555f-46e9-8ebc-1a27a43fb29e","year":2019},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.372760Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:fdc29e10937ce062a2b544312d5ee539546cc48f385f8dc7cd1e91a3ab493441","observation_id":"7247ca58-a1c8-44e2-be46-9b9501f0b73a","resolution":{"observed_at":"2026-08-11T17:24:22.931342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-09T20:34:52.923500Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-11T17:24:22.375953Z","title":"Decoupled weight decay regularization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.375953Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:1e64d6e2f7543f5bd1afd502fac6a8c218828cb7eb24d5f397cbe467bd1eb94f","observation_id":"d24612d3-acd4-4bd7-83f7-68b620742567","resolution":{"observed_at":"2026-08-11T17:24:22.375953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.917559Z","title":"Detecting unknown speech spoofing algorithms with nearest neighbors","venue":null,"work_id":"edc3def6-589d-47fa-b342-de7b15c3a7e0","year":2023},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.379583Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:01dbb1e6c985e12e5021e198da0ea22673bde91e9f3b0b34a775558abda12a1a","observation_id":"d1cbe032-a287-4ec8-a2b6-a2d1015016d0","resolution":{"observed_at":"2026-08-11T17:24:22.921089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.906946Z","title":"Audio forensics from acoustic reverberation","venue":null,"work_id":"da3bab5f-5ae4-4e2c-855d-fb74d98e5483","year":2010},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.382735Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:e169047802086cb7cdf06af90c0382d5e9d73fb9895a593b13fb456cc5daaafb","observation_id":"e20c3cdc-5e1b-4ae5-9e0b-456deb337285","resolution":{"observed_at":"2026-08-11T17:24:22.910773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.386003Z","title":"u ller, Pavel Czempin, Franziska Dieckmann, Adam Froghyar, and Konstantin B \\","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.386003Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:20341f55d467337433225ce7b7917828863908457dc9165fd5f93cf45141afd3","observation_id":"13f0be65-d638-4c33-94d0-8f0810e463d0","resolution":{"observed_at":"2026-08-11T17:24:22.386003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1609.03499","last_updated":"2016-09-19T18:04:35Z","snapshot_observed_at":"2026-08-13T01:24:25.628327Z","submitted_at":"2016-09-12T17:29:40Z","title":"WaveNet: A Generative Model for Raw Audio","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.03499","snapshot_observed_at":"2026-08-11T17:24:22.390076Z","title":"Wavenet: A generative model for raw audio","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.390076Z"},"links":{"cited_paper":"/paper/1609.03499","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:0f7ab861ce675acb2777d4734c76f942d9eacacccf5a214975b7c628b6a04983","observation_id":"5aa73031-419c-4f51-a949-4ed69ad28f61","resolution":{"observed_at":"2026-08-11T17:24:22.390076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.896995Z","title":"From speaker verification to deepfake algorithm recognition: Our learned lessons from add2023 track3","venue":null,"work_id":"ce6b1ba1-e419-4323-9f31-7d74dbff2fa1","year":2022},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.393524Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:7716e2456b41bfb24273f9b345b10e6fb6b77ba030c8d7a6f2766015cad1c18a","observation_id":"41c68639-8e85-4c88-bc94-6f09ffb21ac3","resolution":{"observed_at":"2026-08-11T17:24:22.900673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.886946Z","title":"For: A dataset for synthetic speech detection","venue":null,"work_id":"d7a2536e-5958-41a8-a039-c1e43de0d592","year":2019},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.396645Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:767e10056a5e1e07ae9f7eb54280eafecd8dd9761e7f699a36c23d14e869f8e9","observation_id":"740178de-963f-4ca1-98ba-c95160220cfb","resolution":{"observed_at":"2026-08-11T17:24:22.890518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.877913Z","title":"Transformer language models with lstm-based cross-utterance information representation","venue":null,"work_id":"5d930c2f-d1c9-44e0-a881-82cbae6f29db","year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.399903Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:2fc6630dc99ba192d6ad40decc78b9d686ad65c8e9ee66fa16c6ed876d731836","observation_id":"f75e8faa-f601-473f-8026-326773a02e24","resolution":{"observed_at":"2026-08-11T17:24:22.880964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.869335Z","title":"End-to-end anti-spoofing with rawnet2","venue":null,"work_id":"2a6c1f92-ac92-43b1-8529-bc86b41f709a","year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.403243Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:5bf0e91a0ea03255f8dae4d7c29f5d9b2e922c5dd4f17f882cda4b7910d5d0e3","observation_id":"27138424-c667-4405-9640-5dbb183d9485","resolution":{"observed_at":"2026-08-11T17:24:22.872344Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.05441","last_updated":"2019-04-14T11:38:32Z","snapshot_observed_at":"2026-07-06T07:45:20.666021Z","submitted_at":"2019-04-09T15:12:52Z","title":"ASVspoof 2019: Future Horizons in Spoofed and Fake Audio Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.05441","snapshot_observed_at":"2026-08-11T17:24:22.406661Z","title":"Asvspoof 2019: Future horizons in spoofed and fake audio detection","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.406661Z"},"links":{"cited_paper":"/paper/1904.05441","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:427ef5c29425eebaef11de971f2979f132e9f0f1c3f43e88f7ad8fce831a72a4","observation_id":"9d878dfa-ad6f-43da-80d1-fc138e4b9fa6","resolution":{"observed_at":"2026-08-11T17:24:22.406661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.860141Z","title":"Spoofing detection with dnn and one-class svm for the asvspoof 2015 challenge","venue":null,"work_id":"d921af83-42aa-4649-8183-be95f815fa2f","year":2015},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.410276Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:5c3746395e2d602707f6faa4756469691dd25b356a37b255fdb580d3ed4c7919","observation_id":"daccce08-0b4f-4ed6-a69e-e5415a030458","resolution":{"observed_at":"2026-08-11T17:24:22.863626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1703.10135","last_updated":"2017-04-06T21:20:34Z","snapshot_observed_at":"2026-08-05T15:51:16.043022Z","submitted_at":"2017-03-29T16:55:13Z","title":"Tacotron: Towards End-to-End Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1703.10135","snapshot_observed_at":"2026-08-11T17:24:22.413789Z","title":"Tacotron: A fully end-to-end text-to-speech synthesis model","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.413789Z"},"links":{"cited_paper":"/paper/1703.10135","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:24952fee9c0648fe56b4038797adebeeca84694181db8b900c704a16f3f3f717","observation_id":"331c1ca5-b8d4-42fb-ad89-26a8d430e5ce","resolution":{"observed_at":"2026-08-11T17:24:22.413789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.850407Z","title":"Asvspoof 2019: A large-scale public database of synthesized, converted and replayed speech","venue":null,"work_id":"4fd6925b-11f3-4e32-890b-69539218d484","year":2019},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.416798Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:60ff9aed0a452dcbbf678f99c2fb0a8bd3c8b5da348006bcfdc9fb3f0f754249","observation_id":"212d1436-3813-46ea-a630-f0df7501a24a","resolution":{"observed_at":"2026-08-11T17:24:22.853933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.840817Z","title":"Sas: A speaker verification spoofing database containing diverse attacks","venue":null,"work_id":"99d78d74-b933-4619-909d-4cf13a8accf1","year":2015},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.419534Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:0d1726b37d795a0543ed65b694daadac8695499bd4eb3908e46ee89e8d612907","observation_id":"5737d445-a07f-4063-aad8-0685b67b7d25","resolution":{"observed_at":"2026-08-11T17:24:22.844231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.00537","last_updated":"2021-09-01T16:17:31Z","snapshot_observed_at":"2026-07-06T11:43:28.084184Z","submitted_at":"2021-09-01T16:17:31Z","title":"ASVspoof 2021: accelerating progress in spoofed and deepfake speech detection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.00537","snapshot_observed_at":"2026-08-11T17:24:22.422384Z","title":"Asvspoof 2021: accelerating progress in spoofed and deepfake speech detection","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.422384Z"},"links":{"cited_paper":"/paper/2109.00537","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:8a2004cbd51b22faf551cf1180e67a6a4a53f2b24f20317d333074c2938aa0ca","observation_id":"9b7f3e3f-949f-460d-ad38-484526177200","resolution":{"observed_at":"2026-08-11T17:24:22.422384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.830727Z","title":"Detecting digital audio forgeries by checking frame offsets","venue":null,"work_id":"7e0f2509-615b-48db-8914-72d06ba3bcb3","year":2008},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.425536Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:d9bdc95fbeaf7e7cfbf7f8897d3210aca03d37bb24530d45d11caa5c9186615a","observation_id":"17e9d558-3eaf-4e88-aa08-268e5bab0937","resolution":{"observed_at":"2026-08-11T17:24:22.834575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.820372Z","title":"Revisiting anchor mechanisms for temporal action localization","venue":null,"work_id":"5f449798-f19c-418d-84ed-f8e34b076067","year":2020},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.429716Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:abffbb90afd129248e957489108d85445a3d8567c878b73d4d2e35ba6c19a94a","observation_id":"37b3cb0b-11fb-4a5b-84d9-0cfa8241ef42","resolution":{"observed_at":"2026-08-11T17:24:22.824048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.01051","last_updated":"2021-10-15T22:04:39Z","snapshot_observed_at":"2026-08-10T18:31:07.739445Z","submitted_at":"2021-05-03T17:51:09Z","title":"SUPERB: Speech processing Universal PERformance Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.01051","snapshot_observed_at":"2026-08-11T17:24:22.432903Z","title":"Superb: Speech processing universal performance benchmark","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.432903Z"},"links":{"cited_paper":"/paper/2105.01051","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:9948eab477cb20a114d7054181a23dde5dea7a591c625cf316615b3dec5700ea","observation_id":"aa365929-7583-4da4-924a-e36fdb2f6d70","resolution":{"observed_at":"2026-08-11T17:24:22.432903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03617","last_updated":"2023-12-16T02:17:19Z","snapshot_observed_at":"2026-08-07T05:26:29.050197Z","submitted_at":"2021-04-08T08:57:13Z","title":"Half-Truth: A Partially Fake Audio Detection Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.03617","snapshot_observed_at":"2026-08-11T17:24:22.436246Z","title":"Half-truth: A partially fake audio detection dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.436246Z"},"links":{"cited_paper":"/paper/2104.03617","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:711f5d610aff3e9ac348eb580b82343d3fc49cad48db06a2e9e2248336598113","observation_id":"50f5ed7f-5e8a-4f4c-97f9-3ca5a97327a4","resolution":{"observed_at":"2026-08-11T17:24:22.436246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.809598Z","title":"Add 2022: the first audio deep synthesis detection challenge","venue":null,"work_id":"3b08d8d0-e82a-49cf-bf2d-4558182d41f6","year":2022},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.440011Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:a3035bbf4d1686cfd6be4cd2b5fa7387b8ad3a24c875419166f9cad8e2b33ecb","observation_id":"88ff7d46-976a-4e83-a7f5-e85b33a16063","resolution":{"observed_at":"2026-08-11T17:24:22.813565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13774","last_updated":"2023-05-23T07:42:52Z","snapshot_observed_at":"2026-07-06T15:31:13.189124Z","submitted_at":"2023-05-23T07:42:52Z","title":"ADD 2023: the Second Audio Deepfake Detection Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13774","snapshot_observed_at":"2026-08-11T17:24:22.443432Z","title":"Add 2023: the second audio deepfake detection challenge","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.443432Z"},"links":{"cited_paper":"/paper/2305.13774","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:4b420472d43401e0637ae8bbb2923249b499c75d08bfab5463d22c7dc593c140","observation_id":"3447e132-51ed-41fe-b91f-ecad92a411c6","resolution":{"observed_at":"2026-08-11T17:24:22.443432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.798420Z","title":"Searching multi-rate and multi-modal temporal enhanced networks for gesture recognition","venue":null,"work_id":"cd3b12cb-9a4a-4503-b7d8-dd27cd57ef02","year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.446815Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:31b4c4dafa5c33ebecd613be17706e1d0cd24a1f18781a7b7546cad149c58afc","observation_id":"e8c50d3d-fff2-422c-8966-831579f935fe","resolution":{"observed_at":"2026-08-11T17:24:22.802122Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.787772Z","title":"Deepfake algorithm recognition system with augmented data for add 2023 challenge","venue":null,"work_id":"07678336-0353-45cd-b347-2626ba71bf89","year":2023},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.449823Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:6cae6ef63950343657a0dfd19932be663f7529855edc39e673602e04a9dfa45e","observation_id":"10cbb280-92ae-4260-bb76-48d67327323f","resolution":{"observed_at":"2026-08-11T17:24:22.791485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.02518","last_updated":"2021-06-15T15:41:34Z","snapshot_observed_at":"2026-07-06T10:56:54.885179Z","submitted_at":"2021-04-06T13:52:31Z","title":"An Initial Investigation for Detecting Partially Spoofed Audio","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.02518","snapshot_observed_at":"2026-08-11T17:24:22.453002Z","title":"An initial investigation for detecting partially spoofed audio","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.453002Z"},"links":{"cited_paper":"/paper/2104.02518","citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:597deb1c0d87204c4429078a407e17adf431580b0d2ad19ac8113d27e00eaa0f","observation_id":"e5990417-5834-4d9e-b2d3-208104f4caa0","resolution":{"observed_at":"2026-08-11T17:24:22.453002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.776615Z","title":"Actionformer: Localizing moments of actions with transformers","venue":null,"work_id":"36776816-6e2b-44ea-89c6-9e5c3d4c2802","year":2022},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.456448Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:6ae24d3f07a31cbd65aee5897fda1490e25b6ce2efb91797a5a9e312f2c6f54f","observation_id":"0925d8e0-471d-4749-adcc-24a698e22423","resolution":{"observed_at":"2026-08-11T17:24:22.780612Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.766049Z","title":"Audio recording location identification using acoustic environment signature","venue":null,"work_id":"b88f8de8-a390-46ed-a86b-34f2223c4fd5","year":2013},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.460003Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:1353041114395c7cc331d08e35513f330c4eb06d9e19fbebc503a026792d11fe","observation_id":"395b5225-5a54-4eeb-a774-e2630a09bd58","resolution":{"observed_at":"2026-08-11T17:24:22.769870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T17:24:22.463020Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T17:24:22.463020Z"},"links":{"citing_paper":"/paper/2412.09032"},"observation_digest":"sha256:912888afd785d03ad1e3f4f851e2230f6ffd2c989db2c97b12ed3e020575a555","observation_id":"87e07a23-9dc4-44ff-83e3-10c1c2de7943","resolution":{"observed_at":"2026-08-11T17:24:22.463020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.09032","last_updated":"2025-07-17T03:29:13Z","latest_version":3,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-12T05:28:06.409447Z","submitted_at":"2024-12-12T07:48:17Z","title":"Speech-Forensics: Towards Comprehensive Synthetic Speech Dataset Establishment and Analysis"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":2,"verified_fuzzy":27},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 0 inbound Pith citation observations for arXiv:2412.09032."}