{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FRFOETALFA7B7RQH4EZLDMZXMO","short_pith_number":"pith:FRFOETAL","schema_version":"1.0","canonical_sha256":"2c4ae24c0b283e1fc607e132b1b33763819829800cca44ac0c2f114e790893fc","source":{"kind":"arxiv","id":"2308.14970","version":1},"attestation_state":"computed","paper":{"title":"Audio Deepfake Detection: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Chenglong Wang, Chu Yuan Zhang, Jiangyan Yi, Jianhua Tao, Xiaohui Zhang, Yan Zhao","submitted_at":"2023-08-29T01:50:01Z","abstract_excerpt":"Audio deepfake detection is an emerging active topic. A growing number of literatures have aimed to study deepfake detection algorithms and achieved effective performance, the problem of which is far from being solved. Although there are some review literatures, there has been no comprehensive survey that provides researchers with a systematic overview of these developments with a unified evaluation. Accordingly, in this survey paper, we first highlight the key differences across various types of deepfake audio, then outline and analyse competitions, datasets, features, classifications, and ev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.14970","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-08-29T01:50:01Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"d38987f46203055b376a42a076ccc3d088d054c9da1df30fc8ec9a7729f73f47","abstract_canon_sha256":"3bb6b707511ec7b3b64b6274a5506244ad2f54f9d1823c42c030d8ce90fe9b26"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:45:41.634950Z","signature_b64":"jzFnkP/LxSJMej1ZTyOLblB4pAMMDXuTy/4gpVO8AVILVJwu87XZf3K+a46AY609NQ1scai4ZuCtAjaH7MzbDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c4ae24c0b283e1fc607e132b1b33763819829800cca44ac0c2f114e790893fc","last_reissued_at":"2026-07-05T06:45:41.634539Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:45:41.634539Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio Deepfake Detection: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Chenglong Wang, Chu Yuan Zhang, Jiangyan Yi, Jianhua Tao, Xiaohui Zhang, Yan Zhao","submitted_at":"2023-08-29T01:50:01Z","abstract_excerpt":"Audio deepfake detection is an emerging active topic. A growing number of literatures have aimed to study deepfake detection algorithms and achieved effective performance, the problem of which is far from being solved. Although there are some review literatures, there has been no comprehensive survey that provides researchers with a systematic overview of these developments with a unified evaluation. Accordingly, in this survey paper, we first highlight the key differences across various types of deepfake audio, then outline and analyse competitions, datasets, features, classifications, and ev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.14970","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.14970/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.14970","created_at":"2026-07-05T06:45:41.634591+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.14970v1","created_at":"2026-07-05T06:45:41.634591+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.14970","created_at":"2026-07-05T06:45:41.634591+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRFOETALFA7B","created_at":"2026-07-05T06:45:41.634591+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRFOETALFA7B7RQH","created_at":"2026-07-05T06:45:41.634591+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRFOETAL","created_at":"2026-07-05T06:45:41.634591+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16532","citing_title":"Dual-Granularity Orthogonal Disentanglement for Generalizable Audio Deepfake Detection","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04205","citing_title":"DetectZoo: A Unified Toolkit for AI-Generated Content Detection Across Text, Audio, and Image Modalities","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30791","citing_title":"Probing-Guided Layer Selection from Self-Supervised Speech Models for Generalizable Audio Deepfake Detection","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23201","citing_title":"MixFake: Benchmarking and Enhancing Audio Deepfake Detection in Diverse Real-world Mixed Audio","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20266","citing_title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20799","citing_title":"AuthGlass: Benchmarking Voice Liveness Detection and Authentication on Smart Glasses via Comprehensive Acoustic Features","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24674","citing_title":"Advancing Zero-Shot Open-Set Speech Deepfake Source Tracing","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09007","citing_title":"Gender Fairness in Audio Deepfake Detection: Performance and Disparity Analysis","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04951","citing_title":"Synthetic Trust Attacks: Modeling How Generative AI Manipulates Human Decisions in Social Engineering Fraud","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02913","citing_title":"Split and Conquer Partial Deepfake Speech","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03079","citing_title":"Phoneme-Level Deepfake Detection Across Emotional Conditions Using Self-Supervised Embeddings","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02223","citing_title":"Toward Fine-Grained Speech Inpainting Forensics:A Dataset, Method, and Metric for Multi-Region Tampering Localization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12650","citing_title":"Listening Deepfake Detection: A New Perspective Beyond Speaking-Centric Forgery Analysis","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07241","citing_title":"Asymmetric Phase Coding Audio Watermarking","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13400","citing_title":"Classical Machine Learning Baselines for Deepfake Audio Detection on the Fake-or-Real Dataset","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16254","citing_title":"ArtifactNet: Detecting AI-Generated Music via Forensic Residual Physics","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19949","citing_title":"Indic-CodecFake meets SATYAM: Towards Detecting Neural Audio Codec Synthesized Speech Deepfakes in Indic Languages","ref_index":80,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO","json":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO.json","graph_json":"https://pith.science/api/pith-number/FRFOETALFA7B7RQH4EZLDMZXMO/graph.json","events_json":"https://pith.science/api/pith-number/FRFOETALFA7B7RQH4EZLDMZXMO/events.json","paper":"https://pith.science/paper/FRFOETAL"},"agent_actions":{"view_html":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO","download_json":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO.json","view_paper":"https://pith.science/paper/FRFOETAL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.14970&json=true","fetch_graph":"https://pith.science/api/pith-number/FRFOETALFA7B7RQH4EZLDMZXMO/graph.json","fetch_events":"https://pith.science/api/pith-number/FRFOETALFA7B7RQH4EZLDMZXMO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO/action/storage_attestation","attest_author":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO/action/author_attestation","sign_citation":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO/action/citation_signature","submit_replication":"https://pith.science/pith/FRFOETALFA7B7RQH4EZLDMZXMO/action/replication_record"}},"created_at":"2026-07-05T06:45:41.634591+00:00","updated_at":"2026-07-05T06:45:41.634591+00:00"}