{"as_of":"2026-07-28T19:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4d7f80ecf9d42c9db7a2ff494ea0466fc7c908d19293b60405e9a95a47ba8528","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T13:31:12.802897Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-07-28T06:31:03.373048+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T13:31:12.802897Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-10T13:35:26.556810Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"cited_work":{"arxiv_id":"2604.13229","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.13229","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","venue":"eess.AS","work_id":"d90b1a02-402c-4fd2-ad83-41014b0a798b","year":2026},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2604.13229","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:444f4201bb8a09bfb166dcf541fe5b564f612ea23e033fc0472b2a6275e857fc","observation_id":"9e5a30ec-3594-45ef-bf92-95f36cb98e06","resolution":{"observed_at":"2026-05-10T13:35:26.558280Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2604.13229/citation-record","integrity":"/paper/2604.13229/integrity","json":"/paper/2604.13229/citation-record.json","paper":"/paper/2604.13229"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"cited_work":{"arxiv_id":"2604.13229","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.13229","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","venue":"eess.AS","work_id":"d90b1a02-402c-4fd2-ad83-41014b0a798b","year":2026},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2604.13229","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:444f4201bb8a09bfb166dcf541fe5b564f612ea23e033fc0472b2a6275e857fc","observation_id":"9e5a30ec-3594-45ef-bf92-95f36cb98e06","resolution":{"observed_at":"2026-05-10T13:35:26.558280Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"However, their robustness to emotional and expressive synthetic speech remains limited","venue":null,"work_id":"46046472-32ee-48e9-a965-450bdab7846d","year":null},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:8c2cd5c4f7d453fb2b78dca7533c396792e7a716e4256d93d68d4b57223d79ee","observation_id":"fa8845e3-fb0b-44c2-9a41-6f4a0dfa77b1","resolution":{"observed_at":"2026-05-18T23:42:53.506249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Figure 1 illustrates the over- all architecture","venue":null,"work_id":"7da718f7-72d5-417a-b403-8cfa3d9cb5a4","year":null},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:1e055087a75fffdc511d6ba8cfd7f525e26f68287ca28b8b8fdc6adc6f75ddc0","observation_id":"3bc26c09-d6ba-4234-8986-db625148b079","resolution":{"observed_at":"2026-05-18T23:42:53.515130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dataset.We group data into training, standard bench- marks, and emotional/expressive benchmarks","venue":null,"work_id":"1d6a0eed-ce48-4ab9-8f88-cefb181c092c","year":2019},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:02c172db6744ec03196880654d55bdfe2e5bdeb830d9bf5aec655386078dc5d2","observation_id":"600e0139-d7d6-49d1-9829-5884dd95017e","resolution":{"observed_at":"2026-05-18T23:42:53.523770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"w/o MP- SI","venue":null,"work_id":"f3c85f79-b5fc-4667-9bb8-caac2b23aa00","year":2019},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:ec0769b41c4d0261137333e456dd103ad2e5ea4c43a9ea5a6ecd19f616a746df","observation_id":"edebf2f0-0267-41a5-88d4-af1ef1a0dcc9","resolution":{"observed_at":"2026-05-18T23:42:53.498498Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"373bf3eb-6de9-45fd-8538-c218e69d7f22","year":2019},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:d513324798f6a2df8f23132981b92e4b462ab44ca5e6179e82050578e08e7dfa","observation_id":"1fa00bee-772a-46be-b9eb-61487b2660a1","resolution":{"observed_at":"2026-05-18T23:42:53.534458Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"• The Office of the Director of National Intelligence (ODNI), Intelligence Advanced Research Projects Activity (IARPA), via the ARTS Program, Contract #D2023-2308110001","venue":null,"work_id":"0c0d3631-8804-45bb-b47e-9a03103238d2","year":null},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:cfa2b21a4d83c4da5e58ce7af9ad3efeb89fe7392064a9773331cf9f71dadf62","observation_id":"ac5a5d5e-df16-4948-ba37-5b1902697db2","resolution":{"observed_at":"2026-05-18T23:42:53.531000Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"These tools were not used to generate scientific content, results, experimental designs, anal- yses, or conclusions","venue":null,"work_id":"ca692925-16d3-4acf-99f6-24440fb059dc","year":null},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:b43a6bcf862b2b350fbb86c233eb1eb6a2383622ce6dc2580be84f8f12ee1384","observation_id":"dbc9ff40-8bd3-40f9-abf1-01701a1524b8","resolution":{"observed_at":"2026-05-18T23:42:53.520063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Battling voice spoofing: a review, comparative analysis, and generalizability evaluation of state-of-the-art voice spoofing counter measures","venue":null,"work_id":"0bbaaa50-36bb-4d9d-8037-172e85c05cab","year":2023},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:ab917506de4c07412951e8684b83cc6b50bab9f6d79dfce421f89ca53413dd52","observation_id":"a73089cb-c7d0-4555-9d1d-7e0253229f02","resolution":{"observed_at":"2026-05-18T23:42:53.510845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A survey on speech deep- fake detection","venue":null,"work_id":"660e4729-8085-46e2-91ee-decd881decd6","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:235724998b021fa40199dea4978340020f81bea83b75ca5579f60c7ea4a6329f","observation_id":"928da945-d534-4a2a-b934-cf187dddc1f4","resolution":{"observed_at":"2026-05-18T23:42:53.501779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Can emotion fool anti-spoofing?","venue":null,"work_id":"fb302e7e-4595-4058-9481-05f59065eecb","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:3226a8be5fe2e2350f8c4605d560e1d59cdb92eea459a6a3e33df29f100d230e","observation_id":"ed2ec76b-78e1-48da-9e9f-4ebfd0cfaaa1","resolution":{"observed_at":"2026-05-18T23:42:53.527167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Emofake: An initial dataset for emotion fake audio detection","venue":null,"work_id":"796939ed-50e6-4123-9dc4-3027f41a4309","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:dc8513a352f349f27a84a27fbe6588365ed9a942a8f3077beefd7bdaf532b38f","observation_id":"b59ca0d8-09f4-4174-a8f3-ed06fe72d2b5","resolution":{"observed_at":"2026-05-18T23:41:56.192306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mlaad: The multi- language audio anti-spoofing dataset","venue":null,"work_id":"2a4a334d-2f3a-49cb-8632-5e946c78f8cf","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:5e24f41eac734597a10231e1c6530eec68b3ba8b4cbe98237e2c9ec23c35aaea","observation_id":"2a1bd8ef-ed14-4851-9a14-8de1620b4f74","resolution":{"observed_at":"2026-05-18T23:41:56.196437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2203.16263","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T12:29:52.019001Z","title":"Does audio deep- fake detection generalize?","venue":null,"work_id":"d697e2fd-c9b8-4039-a8ce-ffcf8a0a8f80","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:eb560da6dfd2210573b69fbd38f404ed8d17d256572b773a1d48b0cf599ee10e","observation_id":"242f3cd2-e8e7-46e8-8136-ca5dc167f9bb","resolution":{"observed_at":"2026-05-10T13:35:26.578566Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Asvspoof 2015: Automatic speaker verification spoofing and countermea- sures challenge evaluation plan","venue":null,"work_id":"4fe0ae34-4192-41d1-961a-7e543a109f2b","year":2015},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:34ad1d76ac7b7731e3a8fd059744bf742560ddd06af3f57c7c7259ce3c23cbeb","observation_id":"393d1e49-c82a-4598-a107-2eeab94a2383","resolution":{"observed_at":"2026-05-18T23:41:56.200338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Add 2022: the first audio deep synthe- sis detection challenge","venue":null,"work_id":"93345150-30b5-4ff9-af37-eec3834c2eee","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:12ebfc5cc91469d443b54117b1fc907a4a48c7943376b0671b02b6c85b81c082","observation_id":"b933451b-d9a4-4ba2-ac57-959cd180f315","resolution":{"observed_at":"2026-05-18T23:41:56.185414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13774","last_updated":"2023-05-23T07:42:52Z","snapshot_observed_at":"2026-07-06T15:31:13.189124Z","submitted_at":"2023-05-23T07:42:52Z","title":"ADD 2023: the Second Audio Deepfake Detection Challenge","version":1},"cited_work":{"arxiv_id":"2305.13774","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.13774","snapshot_observed_at":"2026-07-04T02:19:22.853731Z","title":"ADD 2023: the second audio Deepfake detection challenge","venue":null,"work_id":"808718ce-d91e-44e4-9c14-ac5308ad8428","year":2023},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2305.13774","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:9ad2e03bee2890077ec2c3398ea46e82653e094e5645790926981cf0450ba88e","observation_id":"3d7f83b2-38dd-4b09-91cd-cc247db8de29","resolution":{"observed_at":"2026-05-10T13:35:26.576142Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Asvspoof 2019: A large-scale public database of synthesized, converted and replayed speech","venue":null,"work_id":"82565e67-e7aa-40ee-bf81-836c4e2034e7","year":2019},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:a2aa763d59d3b835a4a487188f781ba9f02b33301a0a377407e50f346c3c6bbc","observation_id":"9ca82b15-c04d-498d-a882-61009eb9c5a6","resolution":{"observed_at":"2026-05-18T23:41:56.181517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Asvspoof 2021: Towards spoofed and deepfake speech de- tection in the wild","venue":null,"work_id":"ff954350-a3fe-4cf5-a39e-3a5396d34cd2","year":2021},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:73d99adc090afe84157d721115c0f4139b6b56d8180fff685e16800cd9726047","observation_id":"28c20390-ca0b-425e-abd3-682a2cb18a23","resolution":{"observed_at":"2026-05-18T23:41:56.204343Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08739","last_updated":"2024-08-16T13:37:20Z","snapshot_observed_at":"2026-07-06T19:01:32.237848Z","submitted_at":"2024-08-16T13:37:20Z","title":"ASVspoof 5: Crowdsourced Speech Data, Deepfakes, and Adversarial Attacks at Scale","version":1},"cited_work":{"arxiv_id":"2408.08739","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.08739","snapshot_observed_at":"2026-07-04T07:49:38.746872Z","title":"ASVspoof 5: Crowdsourced speech data at scale","venue":null,"work_id":"00f9ada8-7f80-449e-95d3-b9f58bfad87a","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2408.08739","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:c4a6818361a943f6ee3154a54727bb0bf4003ee66a6d82ce05beec8aedd5d172","observation_id":"bf630327-c7e9-4765-9adb-7790f3137f51","resolution":{"observed_at":"2026-05-10T13:35:26.555145Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.21676","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hula: Prosody-aware anti-spoofing with multi-task learning for expressive and emo- tional synthetic speech","venue":null,"work_id":"bc91ea6e-99f5-46e4-a0f4-3032d46c38b5","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:8fbfcd30dffd075cdfb0f93bca4ae8ee3bb7d59dfeffac79a782cab8b705790b","observation_id":"3be9227f-bb8f-4b04-9eaf-9eed07129820","resolution":{"observed_at":"2026-05-10T13:35:26.581251Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"End-to-end anti-spoofing with rawnet2","venue":null,"work_id":"6b70e290-ca35-486f-8b28-52d089f96e11","year":2021},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:642f75d9d9f9dcd1fe35ef5426ad17ef36b153649aaed1f8938f4ec14efc7f7c","observation_id":"840dd24c-b01e-4bfd-a416-7f56e13ab105","resolution":{"observed_at":"2026-05-18T23:41:56.248236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aasist: Audio anti-spoofing using in- tegrated spectro-temporal graph attention networks","venue":null,"work_id":"356dc461-5832-4bdd-bf23-e94d9bd22f68","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:7e3536a19bb49bd96d273846806b26c6f74130e9bc95b7f08760e33aecfdff9c","observation_id":"b500f070-cda1-4d01-821b-ea1f94b98bcb","resolution":{"observed_at":"2026-05-18T23:41:56.241790Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.09637","last_updated":"2020-09-21T06:38:19Z","snapshot_observed_at":"2026-07-06T09:57:15.968544Z","submitted_at":"2020-09-21T06:38:19Z","title":"Light Convolutional Neural Network with Feature Genuinization for Detection of Synthetic Speech Attacks","version":1},"cited_work":{"arxiv_id":"2009.09637","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2009.09637","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Light convolutional neu- ral network with feature genuinization for detection of synthetic speech attacks","venue":null,"work_id":"64fd0dd8-31e7-478b-8f0e-a412e9527f5f","year":2009},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2009.09637","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:57cb99feb56558bd516d0380a8e5e1da8158114107e076c296786553494d660d","observation_id":"fad7278b-d9c8-4504-a1fc-0d994fb57727","resolution":{"observed_at":"2026-05-10T13:35:26.560856Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.01120","last_updated":"2019-04-01T21:47:00Z","snapshot_observed_at":"2026-07-06T07:43:12.012344Z","submitted_at":"2019-04-01T21:47:00Z","title":"ASSERT: Anti-Spoofing with Squeeze-Excitation and Residual neTworks","version":1},"cited_work":{"arxiv_id":"1904.01120","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1904.01120","snapshot_observed_at":"2026-07-04T23:31:23.938248Z","title":"Assert: Anti- spoofing with squeeze-excitation and residual networks","venue":null,"work_id":"7921f4dc-8264-4014-b2b5-dc2b07e7cfd4","year":1904},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/1904.01120","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:3d393eaa8c1c0269fb0a1cb477d4f4dcde2ae360d4f5bc75da415a366a34ecb2","observation_id":"a2274403-6905-4a7a-9251-299eb359c147","resolution":{"observed_at":"2026-07-04T23:31:23.938248Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Spoof detection using voice contribution on lfcc features and resnet-34","venue":null,"work_id":"ded595dc-4c8b-4bbe-bc65-45728459f92a","year":2023},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:a5dc7e18da6c917bf92e5131878a4c2f5eff1637f49c7664885e21cfb485e0a2","observation_id":"bdc9491d-f8c3-4f5d-a6fe-96feba97124e","resolution":{"observed_at":"2026-05-18T23:41:56.174471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Audio deepfake detection with self-supervised xls-r and sls classifier","venue":null,"work_id":"bae905b5-44f2-4c3a-acea-522b81080843","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:4dd2d0ecf1a08975aafc18fa16caa05d5aa14cfa7760224e2b8df809eec1f347","observation_id":"12161f26-616b-4891-8e62-a1ea1f255381","resolution":{"observed_at":"2026-05-18T23:41:56.167451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.12233","last_updated":"2022-02-28T11:50:57Z","snapshot_observed_at":"2026-07-06T12:41:22.635071Z","submitted_at":"2022-02-24T17:55:00Z","title":"Automatic speaker verification spoofing and deepfake detection using wav2vec 2.0 and data augmentation","version":2},"cited_work":{"arxiv_id":"2202.12233","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2202.12233","snapshot_observed_at":"2026-07-03T03:47:35.550760Z","title":"Automatic speaker verification spoofing and deepfake detection using wav2vec 2.0 and data augmentation","venue":null,"work_id":"4453706c-8f12-4bd1-ad86-c9af276c509e","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2202.12233","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:d7a69cec28b6a5ae208fb0d337e70f1163b9976d29fd437b0819dc9f824f0001","observation_id":"015e6f13-537d-4af5-a3a5-92adc3cc21cc","resolution":{"observed_at":"2026-05-10T13:35:26.573529Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14726","last_updated":"2025-02-20T16:52:55Z","snapshot_observed_at":"2026-07-06T20:39:57.316580Z","submitted_at":"2025-02-20T16:52:55Z","title":"Pitch Imperfect: Detecting Audio Deepfakes Through Acoustic Prosodic Analysis","version":1},"cited_work":{"arxiv_id":"2502.14726","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14726","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pitch imperfect: Detecting audio deepfakes through acoustic prosodic analysis","venue":null,"work_id":"a05f3251-a2dc-40f3-bbaf-97461fc2aa2d","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2502.14726","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:53f28bbeb749654172bda1f884d15e94f5a75d49f088f27d94f3cf7152a10547","observation_id":"671f2b13-b9ff-49f7-a151-da80a7cc0bc6","resolution":{"observed_at":"2026-05-10T13:35:26.568518Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deepfake speech detection through emotion recognition: a semantic approach","venue":null,"work_id":"14fb2f6a-b641-4c8e-a2a9-bc941414008a","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:f96464d1380b5b66915a5f1cbb0b0cfe7c2c9cdc3cf872a1a7f55a9dce1c5433","observation_id":"bbb14465-e56f-461b-a0df-1c215310babf","resolution":{"observed_at":"2026-05-18T23:41:56.208615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Combining automatic speaker verification and prosody analysis for synthetic speech detection","venue":null,"work_id":"cdcf4354-2d93-43f7-a189-16f00651567e","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:8420c90db17038d81ff888f61e6a1b407120d73fc7381a008e6191424ee221eb","observation_id":"57438390-3ff0-4e05-aa76-82f55845c8ee","resolution":{"observed_at":"2026-05-18T23:41:56.224601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.10781","last_updated":"2025-09-13T01:58:34Z","snapshot_observed_at":"2026-07-06T22:29:38.297334Z","submitted_at":"2025-09-13T01:58:34Z","title":"Emoanti: audio anti-deepfake with refined emotion-guided representations","version":1},"cited_work":{"arxiv_id":"2509.10781","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.10781","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Emoanti: au- dio anti-deepfake with refined emotion-guided representations","venue":null,"work_id":"bd1fabac-98da-4b85-b708-a0866885970b","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2509.10781","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:54c0e6cbe060c0190971019d209df6889c793a30b51b115453732cf2a478dc48","observation_id":"01ad08ff-6181-4303-a8bd-0189ccb577c5","resolution":{"observed_at":"2026-05-10T13:35:26.563364Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Finding the human voice in ai: Insights on the perception of ai-voice clones from naturalness and similarity ratings","venue":null,"work_id":"17453bce-dd7c-40d7-b3e5-1409f5c4553c","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:95c7dbdcd4c17140260b9316f1d3fd2742e85b8f371ce07bed1c7de87f3d9d6f","observation_id":"73762340-e2a6-4ecf-b83e-1faa23b50e64","resolution":{"observed_at":"2026-05-18T23:42:53.920779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How do users perceive deepfake personas? investigating the deepfake user perception and its implications for human-computer interaction","venue":null,"work_id":"39a093f7-b4b5-49c3-86ad-8e3432edcd59","year":2023},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:7c84b9434a9810a2b97f8fabb2adf20605290dbfe30f39b0e4f1efc38a4b94fd","observation_id":"0c3a64e5-1e0d-4d29-bc57-c7c0f8c87c2f","resolution":{"observed_at":"2026-05-18T23:41:56.177907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Subjective perception and objective evaluation of speech naturalness for deepfake de- tection","venue":null,"work_id":"0cac26ea-cae3-4f07-afd0-8070a8c1cf84","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:09f6a85b619cc3baed2ef4054ff4aec61ba7c9c7237ea6b5074c0965515c97f3","observation_id":"2140a1b0-8c36-4d10-bc5b-381f17429400","resolution":{"observed_at":"2026-05-18T23:41:56.188964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Slim: Style- linguistics mismatch model for generalized audio deepfake de- tection","venue":null,"work_id":"d80f6baf-6d55-4a79-90d9-2a1836188ae3","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:f0bc075908afca1b3c720c85cc7fcfba2ec1a14c9ee9977c11f41f635528d66c","observation_id":"3ee59924-101a-4b2b-87ff-323db7403872","resolution":{"observed_at":"2026-05-18T23:41:56.212073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xlsr-mamba: A dual-column bidirec- tional state space model for spoofing attack detection","venue":null,"work_id":"9fd3dbba-4179-4c3d-a10c-8148f0832f4e","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:c7dc467b46849d8d461a8f863bedec56dba8286a3bf8e4ae1e83a2df09cbf07b","observation_id":"d5dbffa9-4e17-4953-ab22-09fdd7bae1de","resolution":{"observed_at":"2026-05-18T23:41:56.164218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Emoq-tts: Emo- tion intensity quantization for fine-grained controllable emotional text-to-speech","venue":null,"work_id":"bd0912f0-4708-4c80-9fc6-02997b5d36e9","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:980cf2e640bdd640c0c274be398bb253488c679a14172256ae207770b1bc7643","observation_id":"052237ae-9d05-4f3c-a2d9-3577ec4b9dd0","resolution":{"observed_at":"2026-05-18T23:41:56.171036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Glow-tts: A genera- tive flow for text-to-speech via monotonic alignment search","venue":null,"work_id":"c8580fd6-6097-4e2c-b3ca-3a0b09367dd6","year":2020},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:cf897d82b09895bb5c1872b86733e405c6bafa001b9ce82946078a27340c7e35","observation_id":"b92da716-8065-4bed-a6e7-c43f7f96c450","resolution":{"observed_at":"2026-05-18T23:41:56.227948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-07-06T19:30:26.233621Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:e79aa3bf500aaec1be12aed8789d106333fa180783aa045b681e86e231ad97b1","observation_id":"31716ffd-6c08-48dd-8ef0-3fc45e234fff","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Laugh now cry later: Controlling time-varying emotional states of flow- matching-based zero-shot text-to-speech","venue":null,"work_id":"1b5c962a-baa0-49f1-861d-c63faa64d1eb","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:805d56988a2ebaacf9f67ff3fc151aa827e6369d59b75b9a670fd924c37eedad","observation_id":"b2fb1132-7212-40c9-af15-6d2580d91af1","resolution":{"observed_at":"2026-05-18T23:42:53.917060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Expressive-vc: Highly expressive voice conversion with attention fusion of bottleneck and perturbation features","venue":null,"work_id":"9298bed8-6630-480d-8dbf-f3099d31d22e","year":2023},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:11919b1e1552139050fcbd2023e29808d71fe3fb3adcadb9842e1d926c3c6015","observation_id":"55e3a807-585b-49e7-885c-4ba31c08ebbd","resolution":{"observed_at":"2026-05-18T23:42:53.912626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pmvc: Data augmentation-based prosody modeling for expres- sive voice conversion","venue":null,"work_id":"32f10905-b67b-4f64-8b45-b0f40343ac90","year":2023},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:2ec0d68ff726e1fd19179177bfcb4aa98baf8e163388abea7b811ee97c54a4ce","observation_id":"633a179e-f38f-4f57-8af6-9cbf59eba136","resolution":{"observed_at":"2026-05-18T23:42:53.892948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Emotion intensity and its control for emotional voice conversion","venue":null,"work_id":"0df40835-b304-402e-963c-7334aa78358f","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:e8d78aa7a93dfa62f51d722ceaa1423da77c00a5e3ffd41d71547e7a3a0649a6","observation_id":"777c5018-b016-4396-ae66-aad89045a20b","resolution":{"observed_at":"2026-05-18T23:42:53.913946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T06:40:46.422915Z","title":"Prosody in context: A review","venue":null,"work_id":"abda208a-7c2b-456f-8822-315ae0f31e53","year":2015},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:ef3aa1d82213d65fc6e7af2d451ef40e6e75e3cc99a95a4a52ede4b69eb6daa6","observation_id":"380ef486-5941-4d03-a4a5-6250cd6b7d04","resolution":{"observed_at":"2026-05-18T23:42:53.910436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning second language suprasegmentals: Effect of l2 experience on prosody and fluency characteristics of l2 speech","venue":null,"work_id":"502044dd-057f-4e54-9e06-a163eea578ce","year":2006},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:2ff9b3da09f6ad16a828cd9e4e9ec71b22759d920e099a3a6853d6207812e642","observation_id":"09544bd3-e124-44f6-b076-9a9488216683","resolution":{"observed_at":"2026-05-18T23:41:56.255490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Variation adds to prosodic typology","venue":null,"work_id":"8408483c-d1cb-4014-a89f-2f471bb1893e","year":2002},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:79fb35e4f929141b5a8d47a863dde12790eda8e4a631fc37b87e60b32b3a3cb9","observation_id":"b76087d9-f571-42bf-bc05-57d30c81fa6d","resolution":{"observed_at":"2026-05-18T23:41:56.245021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"De- tection of cross-dataset fake audio based on prosodic and pronun- ciation features","venue":null,"work_id":"1e023727-06f5-4fe4-b99a-5eda649fb972","year":2023},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:eddb8b3be9aa51f261c6d3e1e966853fcca6e52dc4216451e1e68d22f1fe1466","observation_id":"3d969383-0453-441d-9a95-354b1554956f","resolution":{"observed_at":"2026-05-18T23:41:56.258289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Investigating voiced and unvoiced regions of speech for audio deepfake detection","venue":null,"work_id":"de189740-0010-4205-a899-0020fddeb752","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:d14ee7e1dd6cf4118aeccd2d7c7296b6dc4cb8a7e9900cd7038512182391e5ea","observation_id":"5102fd9d-21e6-4e6f-b8d6-da0725afe8e6","resolution":{"observed_at":"2026-05-18T23:41:56.231335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.07143","last_updated":"2020-08-10T13:50:24Z","snapshot_observed_at":"2026-07-06T09:20:26.283047Z","submitted_at":"2020-05-14T17:02:15Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification","version":3},"cited_work":{"arxiv_id":"2005.07143","doi":"10.48550/arxiv.2005.07143","metadata_source":"pith","pith_arxiv_id":"2005.07143","snapshot_observed_at":"2026-07-10T12:57:07.595481Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification","venue":"eess.AS","work_id":"45c323c0-4c7b-4c68-9f61-82af9cce05aa","year":2020},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2005.07143","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:d861bb2cd52c4052ca3c104aa3408071f25bc175461094a4524083e9255cb64f","observation_id":"ec33642e-90b0-4553-81f2-ef65d3ca2879","resolution":{"observed_at":"2026-05-10T13:35:26.583962Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02584","last_updated":"2025-06-03T08:04:03Z","snapshot_observed_at":"2026-07-06T21:35:36.836386Z","submitted_at":"2025-06-03T08:04:03Z","title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","version":1},"cited_work":{"arxiv_id":"2506.02584","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02584","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prosodic struc- ture beyond lexical content: A study of self-supervised learning","venue":null,"work_id":"044c90e4-70d5-4951-bb6d-f93ca4fc2ae1","year":2025},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2506.02584","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:7456d4a3149e5c188a52a766487f62a94cd6c29bad75b8d745c9eb516ac4cd3c","observation_id":"d7cc8f92-7129-4540-b693-18d5efae5d6b","resolution":{"observed_at":"2026-05-10T13:35:26.588829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T12:32:25.940177Z","title":"Lib- rispeech: An asr corpus based on public domain audio books","venue":null,"work_id":"4dc39a57-144b-4c81-a4a8-a17b1356e24a","year":2015},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:da8f1b76646b2a0025799ec2abb4f85017564a8dc4285cd1cc909090066ce3ef","observation_id":"a20fb279-f4d5-44e9-96f6-7e1bef465366","resolution":{"observed_at":"2026-05-18T23:41:56.234450Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","venue":null,"work_id":"327df064-83c3-4dc9-b81a-9ad35840ac43","year":2021},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:6a441c5c8bfebd880370991ffc3a51e55b4f381f65aafea1ac5ba6592e995456","observation_id":"5c22e4d0-4c19-4604-aa1f-f28ffa50fee2","resolution":{"observed_at":"2026-05-18T23:41:56.238562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04904","last_updated":"2024-06-07T12:56:11Z","snapshot_observed_at":"2026-07-06T18:27:07.473039Z","submitted_at":"2024-06-07T12:56:11Z","title":"XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model","version":1},"cited_work":{"arxiv_id":"2406.04904","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04904","snapshot_observed_at":"2026-07-08T18:15:21.439861Z","title":"Xtts: a massively mul- tilingual zero-shot text-to-speech model","venue":"eess.AS","work_id":"0fdda5f3-0b91-4735-8335-b150db16d2b8","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2406.04904","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:0791e3fa52cf248b2fd3f93d1d6b70934a23259f740b296cbfc1406e6dba8076","observation_id":"00793784-0913-4347-8e8e-541819072528","resolution":{"observed_at":"2026-05-10T13:35:26.593634Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fastpitch: Parallel text-to-speech with pitch pre- diction","venue":null,"work_id":"baae124c-6693-465d-99e8-cd0030d4f787","year":2021},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:da72707cd09141893d4ce12e941f6064c77771e42c9562acc51091ff68fa8e3e","observation_id":"0b696d15-adca-4e7b-98ae-2e8e87af26f6","resolution":{"observed_at":"2026-05-18T23:41:56.251350Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Grad-tts: A diffusion probabilistic model for text-to-speech","venue":null,"work_id":"607ce18d-3aa7-4407-b2e0-4426fc3ea4f8","year":2021},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:a0fa5b4e64b629de23737b7a0d1b145a2cb270b7ba52dcd526937c9bcd55a6f5","observation_id":"75f0f241-0cf5-4b66-8675-6f332dbaf7d8","resolution":{"observed_at":"2026-05-18T23:42:53.907462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.10394","last_updated":"2021-07-23T01:08:09Z","snapshot_observed_at":"2026-07-06T11:31:23.057674Z","submitted_at":"2021-07-21T23:44:17Z","title":"StarGANv2-VC: A Diverse, Unsupervised, Non-parallel Framework for Natural-Sounding Voice Conversion","version":2},"cited_work":{"arxiv_id":"2107.10394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2107.10394","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Starganv2-vc: A diverse, unsupervised, non-parallel framework for natural-sounding voice conversion","venue":null,"work_id":"2a36876c-1b8f-4a56-b224-caa8f5a75eae","year":2021},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2107.10394","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:5d57198b5a24739bf80ec2c13bc9c6a10c0eb24838eac2573c3843455df0b3e0","observation_id":"bc491533-8746-43ae-8f01-dc89efd8725c","resolution":{"observed_at":"2026-05-10T13:35:26.591203Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.13821","last_updated":"2022-08-04T10:25:40Z","snapshot_observed_at":"2026-07-06T11:52:15.682202Z","submitted_at":"2021-09-28T15:48:22Z","title":"Diffusion-Based Voice Conversion with Fast Maximum Likelihood Sampling Scheme","version":2},"cited_work":{"arxiv_id":"2109.13821","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.13821","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Diffusion-based voice conversion with fast maximum likelihood sampling scheme","venue":null,"work_id":"b98a9957-671e-40c3-b314-a143f34ed660","year":2021},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2109.13821","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:5e7f5c97812a912e393b00725e276b43e99028b66aa4552b466ef946d0dac2f9","observation_id":"34b47704-6210-4aa4-804a-daf81940ccf7","resolution":{"observed_at":"2026-05-10T13:35:26.586414Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V oice conversion using speech-to- speech neuro-style transfer","venue":null,"work_id":"5c05b586-4847-410d-8c69-d6014ebbcace","year":2020},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:543abccf9327271cbec557a48b22b6cdb80a453fa5ce1ac508bd8a5732a0a145","observation_id":"38084b70-c446-401f-9eac-34132e3ef105","resolution":{"observed_at":"2026-05-18T23:41:56.215566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Raw- boost: A raw data boosting and augmentation method applied to automatic speaker verification anti-spoofing","venue":null,"work_id":"8e4dc515-e611-4860-85bd-5800c05dfd6e","year":2022},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:58238a6c14a53f4554dca866403ddcaddc06bc17973baa573bfc9e0eb4bd62ff","observation_id":"2ff81406-0527-4ac5-b4f9-22bc14076e05","resolution":{"observed_at":"2026-05-18T23:41:56.218897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-28T06:31:03.373048+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-07-06T23:01:14.824672Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":15,"verified_fuzzy":43},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-07-28T06:31:03.373048+00:00","source":"crossref"},{"observed_at":"2026-07-28T06:30:57.601408+00:00","source":"retraction_watch"}],"thesis":"As of 28 July 2026, this Paper Citation Record lists 60 of 60 outbound references and 1 inbound Pith citation observation for arXiv:2604.13229."}