{"as_of":"2026-08-11T20:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:aacb92e03249a4808e1ef604b7ca0d1ec6f39b3b7eb46c81ec4f200ad6b7c069","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T13:55:19.657537Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T11:50:21.120714Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T13:55:20.441277Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"cited_work":{"arxiv_id":"2509.00186","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.00186","snapshot_observed_at":"2026-08-05T13:55:20.441277Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","venue":"cs.SD","work_id":"cb0884d3-eb96-434a-b710-ca75a3e2e4f5","year":2025},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.355107Z"},"links":{"cited_paper":"/paper/2509.00186","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:8cc4a3a060ee2ddcfe4fe6aa2edd6bbec4224a195aebab03f802ceba32dd1292","observation_id":"539820be-08c3-4cfc-9e72-11de312d9131","resolution":{"observed_at":"2026-08-05T13:55:20.496504Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.00186","snapshot_observed_at":"2026-08-08T11:50:21.120714Z","title":"arXiv preprint arXiv:2509.00186 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05507","last_updated":"2026-08-06T01:19:35Z","snapshot_observed_at":"2026-08-09T23:10:58.519642Z","submitted_at":"2026-08-06T01:19:35Z","title":"AffectDF: The Most Comprehensive Benchmark for Speech Deepfake Detection against Emotionally Expressive Attacks","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-08T11:50:21.120714Z"},"links":{"cited_paper":"/paper/2509.00186","citing_paper":"/paper/2608.05507"},"observation_digest":"sha256:256ccfce922e636bf1d633e8de3b1c502842a3222427c5868d1b14a529d65315","observation_id":"d3aa8b9d-dd17-4099-b745-ff8498d19f03","resolution":{"observed_at":"2026-08-08T11:50:21.120714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2509.00186/citation-record","integrity":"/paper/2509.00186/integrity","json":"/paper/2509.00186/citation-record.json","paper":"/paper/2509.00186"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.642283Z","title":"In the Wild","venue":null,"work_id":"f7ba69ad-367a-45b1-b6ed-89222f95b6c8","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.210708Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:80a38c18b2053daa740cf6cf47c5974c7a392bbb71e4548513fe0940601e08a1","observation_id":"69896137-6f92-4866-b7d0-4dd1e8145067","resolution":{"observed_at":"2026-08-05T13:55:25.733276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.459677Z","title":"Initially, the input audio waveform is chunked into frames and each chunk is processed through TRILL or TRILLs- son models to extract audio representations","venue":null,"work_id":"80a92ba1-efe6-4512-a081-77a8bedc650d","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.438087Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:40cfb40eedd6835635a852fa7044c113f4bb451ce27b37ddfe53b102fd72700d","observation_id":"24dc9567-986e-41ab-98fc-8fe2e3c9c445","resolution":{"observed_at":"2026-08-05T13:55:25.555041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.281077Z","title":"Datasets W e conducted extensive experiments using four distinct English datasets","venue":null,"work_id":"c5dcc74a-698a-4208-89ca-9e9a395ae80f","year":2019},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.535619Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:91c4aa71f6bf92ed222114cee384bcf6262c8aa18956319fc966bed9cf6a0d0d","observation_id":"12dc41d1-7d25-4fa1-b5e9-de1987b5012c","resolution":{"observed_at":"2026-08-05T13:55:25.351331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.035556Z","title":"The results are presented in T able 1","venue":null,"work_id":"58dd1849-7726-4d80-abfc-8b01b28ac91e","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.601890Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:0eb2d27be7fedfadc7a4834e6713fb91b7e31b64edcb464b53a795cfe17e4c7c","observation_id":"dbacd509-8b14-47f6-9841-fdaa09f9df58","resolution":{"observed_at":"2026-08-05T13:55:25.138367Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.712983Z","title":"W e perform extensive experiments to find the most suitable TRILLsson model and optimal chunking duration to balance the local and global temporal features","venue":null,"work_id":"f32aa410-5145-4ac7-af76-799445d33640","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.819089Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:bc1cd300545d208b5fc56d2e43745d6c7c75e63031cd698ad0b916a7ef31782c","observation_id":"52c2db8e-d509-40f9-ab54-82d430059eeb","resolution":{"observed_at":"2026-08-05T13:55:24.776420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.533946Z","title":null,"venue":null,"work_id":"fb256368-602e-4c1d-a6ab-71ed435ff10d","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.895836Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:af46c111e9a6b4271a6716eac3d13e72bf1317e2c442bd11b1ee2f828bccfc48","observation_id":"e1217572-2282-4a80-82cf-c252e5ab2875","resolution":{"observed_at":"2026-08-05T13:55:24.609030Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.583876Z","title":"Relative phase information for detecting human speech and spoofed speech","venue":null,"work_id":"3fcb3bcf-fcda-4a31-a7b4-a0ef366945bf","year":2015},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.468369Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:685c9236d62787d16cbaaa9dfe0cf7d45ab64fb4d0491b35eb520a889b9f23ad","observation_id":"03d5bfdc-5248-4e35-975c-9eecea3681c7","resolution":{"observed_at":"2026-08-05T13:55:23.684648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"cited_work":{"arxiv_id":"2509.00186","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.00186","snapshot_observed_at":"2026-08-05T13:55:20.441277Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","venue":"cs.SD","work_id":"cb0884d3-eb96-434a-b710-ca75a3e2e4f5","year":2025},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.355107Z"},"links":{"cited_paper":"/paper/2509.00186","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:8cc4a3a060ee2ddcfe4fe6aa2edd6bbec4224a195aebab03f802ceba32dd1292","observation_id":"539820be-08c3-4cfc-9e72-11de312d9131","resolution":{"observed_at":"2026-08-05T13:55:20.496504Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.366238Z","title":"ASVspoof 2019: A large-scale public database of synthesized, converted and replayed speech,","venue":null,"work_id":"d090d6c9-e6c4-44d3-bbe6-a371a6f986d0","year":2019},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.971530Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:e08c0368e4d3f6a087c98b1e929fec5862d5b9b5227d65aa159ae53a3fb3d850","observation_id":"ad9bf454-b274-4279-9ca1-b4ca27ac1e9c","resolution":{"observed_at":"2026-08-05T13:55:24.436508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:17.049614Z","title":"Does audio deepfake detection generalize?","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.049614Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:c576307618b35ddc1188b7ad3718ed771ef621095033790d188c544fab9f5e9e","observation_id":"1c65d7c4-6307-432c-a9a0-aeb6b434c17a","resolution":{"observed_at":"2026-08-05T13:55:17.049614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.132691Z","title":"Resnet and model fusion for automatic spoofing detection","venue":null,"work_id":"cd09c863-ee66-4d24-acbe-1ab410d5c25a","year":2017},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.130519Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:131eb90c5ba1e18f436d99529118c505b4630b7c7a2b447363dc2557dcae0609","observation_id":"e061d810-75a2-480f-a57a-cdce569056ac","resolution":{"observed_at":"2026-08-05T13:55:24.232885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.953319Z","title":"Replay and synthetic speech detection with res2net architecture,","venue":null,"work_id":"992df686-a9ae-42c2-a7c2-d88a56b9d1ef","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.201712Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:a69549dbff76232d33fc83f864f50280ddac6273c9382720a88be7ef3f0a78e8","observation_id":"67640e66-052f-4740-8443-76367753ba39","resolution":{"observed_at":"2026-08-05T13:55:24.033733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.757064Z","title":"Fake speech detection using residual network with transformer encoder,","venue":null,"work_id":"955dd1f3-2fb4-49b4-af06-79674c299c35","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.276137Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:92fd9a70c130ef63a895804fb3e4e2f78f34255832abb7d46e35be92b991e9d6","observation_id":"7c1632a8-6075-4c16-a8aa-95ffd633ceb5","resolution":{"observed_at":"2026-08-05T13:55:23.848346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14970","last_updated":"2023-08-29T01:50:01Z","snapshot_observed_at":"2026-07-06T16:11:31.948429Z","submitted_at":"2023-08-29T01:50:01Z","title":"Audio Deepfake Detection: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14970","snapshot_observed_at":"2026-08-05T13:55:17.373586Z","title":"Audio deepfake detection: A survey,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.373586Z"},"links":{"cited_paper":"/paper/2308.14970","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:652ed34f1303bdce79bf18b19e9a1f5a06f8df638ffa337e278b9a7d0c01043a","observation_id":"bfffb3f5-1a15-40f7-8490-85f8460c9173","resolution":{"observed_at":"2026-08-05T13:55:17.373586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:22.189781Z","title":"AASIST: Audio anti-spoofing using integrated spectro-temporal graph attention networks,","venue":null,"work_id":"5ac58a41-eabb-4726-8b00-adb0e63f42d6","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.132951Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:dae3e33d8e0e719a568608734904eb74379a24faa0fd447625fb1e2ca822dfd9","observation_id":"78f7892f-3096-407b-a526-7d0f98aa1e50","resolution":{"observed_at":"2026-08-05T13:55:22.327457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2107.12018","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.183514Z","title":"Ur channel-robust synthetic speech detection system for asvspoof 2021,","venue":null,"work_id":"e483e3ab-f0ce-4740-ace0-8bb6513f4acb","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.535922Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:6ef55598cdc489f3012926555b659c045480b0690875e5dab1a1adc5d3bb64f7","observation_id":"acb983c4-7bea-41d7-914e-729dabfc3ecf","resolution":{"observed_at":"2026-08-05T13:55:20.240050Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.358186Z","title":"A comparison of features for synthetic speech detection,","venue":null,"work_id":"985a810e-253e-491d-a61e-a419a7366b2c","year":2015},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.671294Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:75ae3f33405584ab4f10998a7225e7c7a3dbe99e50663a67c77cb2db8ec20221","observation_id":"82df6cba-b3bb-4d3f-98a8-af410fd48480","resolution":{"observed_at":"2026-08-05T13:55:23.510119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.084331Z","title":"T owards end-to-end synthetic speech detection,","venue":null,"work_id":"911c99e5-97c6-4cc0-9f78-324b31123b7b","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.729282Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:8b400ea5906a878cae37099c270f0e3200136b07a1b2abb2ed77973837b75378","observation_id":"9b35c735-b1f7-4ede-af1e-003e90059020","resolution":{"observed_at":"2026-08-05T13:55:23.181296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:22.754113Z","title":"Multi-task learning improves synthetic speech detection,","venue":null,"work_id":"6960c8dd-46e5-4615-893d-780394fbd59c","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.821025Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:2d117dc64a9623b8ed60d2a9f0ff3579dc67ad52559be0ac9ecd92c491448f67","observation_id":"9bb41311-c237-4418-a328-f622d693e473","resolution":{"observed_at":"2026-08-05T13:55:22.906864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.12710","last_updated":"2021-08-23T17:06:53Z","snapshot_observed_at":"2026-08-10T23:43:16.034667Z","submitted_at":"2021-07-27T10:11:41Z","title":"End-to-End Spectro-Temporal Graph Attention Networks for Speaker Verification Anti-Spoofing and Speech Deepfake Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.12710","snapshot_observed_at":"2026-08-05T13:55:17.884576Z","title":"End-to-end spectro-temporal graph attention networks for speaker verification anti-spoofing and speech deepfake detection,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.884576Z"},"links":{"cited_paper":"/paper/2107.12710","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:5f1201e0bb7793cc5b22ea07b1b1a9350ec3d98f906579ce259b2fee68c98617","observation_id":"0cf6da84-779d-4589-8cf6-a1376b3101e1","resolution":{"observed_at":"2026-08-05T13:55:17.884576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:17.966404Z","title":"Speaker recognition from raw waveform with sincnet,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.966404Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:ce9cb31fda3764a5a907d1eddd9e1ae0fce25471a10f19e1ffaa9ab3f528990f","observation_id":"2870a2f9-4a44-4396-9947-e5fb20705d84","resolution":{"observed_at":"2026-08-05T13:55:17.966404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:22.466078Z","title":"Advanced rawnet2 with attention- based channel masking for synthetic speech detection,","venue":null,"work_id":"569092cb-caad-46c8-a6ba-4d89a60e1cb5","year":2023},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.060418Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:06158adb10df56d394aa2a6b95cef008d0113d2ce5b32c70c41228261387d8d5","observation_id":"1430422c-6d30-4013-8b77-0a4f5d16045b","resolution":{"observed_at":"2026-08-05T13:55:22.606028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.12764","last_updated":"2020-08-06T04:53:37Z","snapshot_observed_at":"2026-08-11T06:28:46.409000Z","submitted_at":"2020-02-25T21:38:24Z","title":"Towards Learning a Universal Non-Semantic Representation of Speech","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.12764","snapshot_observed_at":"2026-08-05T13:55:18.816682Z","title":"T owards learning a universal non-semantic representation of speech,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.816682Z"},"links":{"cited_paper":"/paper/2002.12764","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:e5560a432bcc5ccb10543f4acd054734c0d33df7ccdcaec3d3b1d3e32a794c1c","observation_id":"0944ebae-b4b0-447c-9692-79e4e917056f","resolution":{"observed_at":"2026-08-05T13:55:18.816682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.925690Z","title":"A robust audio deepfake detection system via multi-view feature,","venue":null,"work_id":"423f9061-c57f-49d6-9c83-f70695f3d19b","year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.190662Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:f78dd6944d8ff0db06d98a78eff700890c03eb4c219fa072fec9084907259606","observation_id":"63d176cf-b1fb-445c-a424-4952d9f97d72","resolution":{"observed_at":"2026-08-05T13:55:22.052308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T13:55:18.277810Z","title":"Xls-r: Self-supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.277810Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:0c3f01f8680b5b9af52ae73e5efec11d7528a43f389ba430e031c5aa58c9279c","observation_id":"b1e5f998-c1c8-4785-91e5-dd8b66b957c6","resolution":{"observed_at":"2026-08-05T13:55:18.277810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.690397Z","title":"W avLM: Large-scale self-supervised pre-training for full stack speech processing,","venue":null,"work_id":"33d74146-88ab-4db9-96cc-29efa0c7861a","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.361581Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:266e1ea53359cdb7122f323f559863656e9df1efc37287eb2916465c30f37f5e","observation_id":"624e3ae5-2f4a-48f8-923b-1d257798571b","resolution":{"observed_at":"2026-08-05T13:55:21.827909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.450193Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"f99ea37b-8376-4827-b9d6-4e23b5b01ec3","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.467070Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:dfa9b9555da07c959245d97d82d4b994fac8e13d0ba70e8327bbdc25d3143c91","observation_id":"a4f6b350-65ae-4041-86d2-51f59f3be15c","resolution":{"observed_at":"2026-08-05T13:55:21.602155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.226479Z","title":"Exploring generalization to unseen audio data for spoof- ing: Insights from ssl models,","venue":null,"work_id":"c5d82e27-5268-4d15-b03d-c62648cf1002","year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.555336Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:cb2c23fca4352db97317b4c12214366a4e3eee5bb76ab6a52a8f5f78aa9ccdb9","observation_id":"c79c5911-4407-4dbc-958a-1adc6118ac81","resolution":{"observed_at":"2026-08-05T13:55:21.326332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.07143","last_updated":"2020-08-10T13:50:24Z","snapshot_observed_at":"2026-08-07T15:21:18.262728Z","submitted_at":"2020-05-14T17:02:15Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.07143","snapshot_observed_at":"2026-08-05T13:55:18.642650Z","title":"Ecapa-tdnn: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.642650Z"},"links":{"cited_paper":"/paper/2005.07143","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:803a71156d84071e5d7edd829cf811260c40a8fe6d87b17a9b64353fea31eccd","observation_id":"99ef9ae6-dc3b-4e3b-8422-a1ff46782b01","resolution":{"observed_at":"2026-08-05T13:55:18.642650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.09512","last_updated":"2026-05-18T13:28:38Z","snapshot_observed_at":"2026-07-06T17:17:03.447619Z","submitted_at":"2024-01-17T15:09:02Z","title":"MLAAD: The Multi-Language Audio Anti-Spoofing Dataset","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.09512","snapshot_observed_at":"2026-08-05T13:55:18.739632Z","title":"MLAAD: The multi-language audio anti-spoofing dataset,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.739632Z"},"links":{"cited_paper":"/paper/2401.09512","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:7474166d4f09e170b72b63a4ba74cea041921acb94a8a6c9faf9b6877f9c83e2","observation_id":"e25cfc55-f110-4080-aefb-f4ff357bbc0a","resolution":{"observed_at":"2026-08-05T13:55:18.739632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.661889Z","title":"Audio deepfake detection with self-supervised xls-r and sls classifier,","venue":null,"work_id":"2008256f-8c8f-4ae6-9196-074add1ed88b","year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.455717Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:dff1e8572afe70d6448f1d713fdd6733a39b58e15762ddcb9a76ec99b9a5721e","observation_id":"0c57f54a-865a-4372-b2bd-7ea874dddeec","resolution":{"observed_at":"2026-08-05T13:55:20.719919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.871664Z","title":"This gives us insight that features extracted over a duration of average syllable length are more beneficial for spoofing detection tasks","venue":null,"work_id":"094f8bf3-3a81-4006-ae9d-81c2b8d5f988","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.685798Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:816a56f750a1c23599ae27f254bee60e61990dd7118ebc975b39609f2a3b70a5","observation_id":"cf2c4358-cce5-4475-836e-1ce2b613bb92","resolution":{"observed_at":"2026-08-05T13:55:24.939678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.00236","last_updated":"2022-03-20T21:13:37Z","snapshot_observed_at":"2026-08-02T20:59:20.503566Z","submitted_at":"2022-03-01T05:22:57Z","title":"TRILLsson: Distilled Universal Paralinguistic Speech Representations","version":2},"cited_work":{"arxiv_id":"2203.00236","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.00236","snapshot_observed_at":"2026-08-05T13:55:19.919132Z","title":"TRILLsson: Distilled Universal Paralinguistic Speech Representations","venue":"eess.AS","work_id":"2667345a-4afe-4d53-bbe0-1b2bfb474257","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.869944Z"},"links":{"cited_paper":"/paper/2203.00236","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:8c26009b9aeba2aa35e4effa8d22afcfe7a5a8303c3c044fe66c8885f253c1cd","observation_id":"572231c0-dbee-430e-8b47-9a111b99a5f7","resolution":{"observed_at":"2026-08-05T13:55:19.972447Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.087789Z","title":"Self- normalizing neural networks,","venue":null,"work_id":"4d39ea79-090f-4813-8d93-a3a093d29213","year":2017},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.957534Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:c48c37a2af1c7af498390170292cc835052aaf1a09dba085fd493ee81ad0e982","observation_id":"fc7f36cd-3d29-4d44-ab0b-1efb3dab4bcb","resolution":{"observed_at":"2026-08-05T13:55:21.146141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.014886Z","title":"CSTR VCTK Corpus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92),","venue":null,"work_id":"9c1b14b9-bb98-481d-a4ec-d06f8589559e","year":2019},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.066896Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:d08beb308a032d4d8090254cfd9bcedf6eff0b7e7efd047f305e6de6a8c6e1b7","observation_id":"c6e03aa8-a717-4527-a7f9-b45650198a11","resolution":{"observed_at":"2026-08-05T13:55:21.034346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.918794Z","title":"ASVspoof 2021: T owards spoofed and deepfake speech detection in the wild,","venue":null,"work_id":"f6a3e4fc-2e76-4356-a23d-cba450719c37","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.131502Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:6e8ddab6e1cd2e47a1f35c2f045a34a10b8835fa33920c06e15f4f915ab94f4a","observation_id":"bf1a2af7-3b66-4177-aa54-39c323f20fd6","resolution":{"observed_at":"2026-08-05T13:55:20.963937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.07725","last_updated":"2022-02-04T13:25:23Z","snapshot_observed_at":"2026-08-03T12:01:20.906827Z","submitted_at":"2021-11-15T12:52:50Z","title":"Investigating self-supervised front ends for speech spoofing countermeasures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.07725","snapshot_observed_at":"2026-08-05T13:55:19.205724Z","title":"Investigating self-supervised front ends for speech spoofing countermeasures,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.205724Z"},"links":{"cited_paper":"/paper/2111.07725","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:454cf10fc1c3500d271703c6ce371a3af252731088161b8e3b195a1df257eeea","observation_id":"d88ac0a8-0e34-4277-8ab3-641c88959888","resolution":{"observed_at":"2026-08-05T13:55:19.205724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.776555Z","title":"Lever- aging positional-related local-global dependency for synthetic speech detection,","venue":null,"work_id":"26ff4d59-dc22-44a3-8f85-b6c908e7bd47","year":2023},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.269691Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:a4b41f9668efa92f27085310b21a9ea7f435689ac50682c67ab3efe9f75b4e1f","observation_id":"4f3ef205-f176-42e6-bb65-116e874b1ad4","resolution":{"observed_at":"2026-08-05T13:55:20.848923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13495","last_updated":"2024-10-31T09:11:37Z","snapshot_observed_at":"2026-08-10T13:10:31.818500Z","submitted_at":"2024-06-19T12:35:02Z","title":"DF40: Toward Next-Generation Deepfake Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13495","snapshot_observed_at":"2026-08-05T13:55:19.374122Z","title":"Df40: T oward next-generation deepfake detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.374122Z"},"links":{"cited_paper":"/paper/2406.13495","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:4f075eb84c5bd8e9d910db4b63547a0330c4ee2569fdffffde8be83594aebd62","observation_id":"4f7c1ed3-d8e2-491a-bf7a-1384124f87e6","resolution":{"observed_at":"2026-08-05T13:55:19.374122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.03812","last_updated":"2022-03-10T01:43:15Z","snapshot_observed_at":"2026-07-06T12:45:21.570121Z","submitted_at":"2022-03-08T02:22:28Z","title":"SpeechFormer: A Hierarchical Efficient Framework Incorporating the Characteristics of Speech","version":2},"cited_work":{"arxiv_id":"2203.03812","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.03812","snapshot_observed_at":"2026-08-05T13:55:19.763491Z","title":"SpeechFormer: A Hierarchical Efficient Framework Incorporating the Characteristics of Speech","venue":"cs.SD","work_id":"68bbb993-507f-44cd-bd1e-a34239a0038a","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.542747Z"},"links":{"cited_paper":"/paper/2203.03812","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:7e4eab57f175b1f643b0f1c48412f148b0f848fd9bec2c0ef4e8451406b48eb4","observation_id":"7a2a2976-620d-49fe-b59d-102ed3645829","resolution":{"observed_at":"2026-08-05T13:55:19.811280Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.548738Z","title":"Syllable-level duration determination","venue":null,"work_id":"1bdd1dd2-7e1e-4417-9835-53b43613523f","year":1989},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.657537Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:48a9aa239e9e4a1ddf6b1e30224e1daa9100c8e649599b29b1eebca3e4ffdc29","observation_id":"451cc3be-3a1c-4e1d-ad45-49537757995c","resolution":{"observed_at":"2026-08-05T13:55:20.611212Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":11,"verified_exact":3,"verified_fuzzy":25},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 2 inbound Pith citation observations for arXiv:2509.00186."}