{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XDOUOI4XD5NAMQSG3GKGJO6DX6","short_pith_number":"pith:XDOUOI4X","schema_version":"1.0","canonical_sha256":"b8dd4723971f5a064246d99464bbc3bf82555832afaeafdfd1501fb5b7eb12ee","source":{"kind":"arxiv","id":"2408.08739","version":1},"attestation_state":"computed","paper":{"title":"ASVspoof 5: Crowdsourced Speech Data, Deepfakes, and Adversarial Attacks at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"eess.AS","authors_text":"Hector Delgado, Hemlata Tak, Hye-Jin Shim, Ivan Kukanov, Jee-weon Jung, Junichi Yamagishi, Kong Aik Lee, Massimiliano Todisco, Md Sahidullah, Nicholas Evans, Tomi Kinnunen, Xin Wang, Xuechen Liu","submitted_at":"2024-08-16T13:37:20Z","abstract_excerpt":"ASVspoof 5 is the fifth edition in a series of challenges that promote the study of speech spoofing and deepfake attacks, and the design of detection solutions. Compared to previous challenges, the ASVspoof 5 database is built from crowdsourced data collected from a vastly greater number of speakers in diverse acoustic conditions. Attacks, also crowdsourced, are generated and tested using surrogate detection models, while adversarial attacks are incorporated for the first time. New metrics support the evaluation of spoofing-robust automatic speaker verification (SASV) as well as stand-alone de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.08739","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-08-16T13:37:20Z","cross_cats_sorted":["cs.AI","cs.SD"],"title_canon_sha256":"aab72e052d7056a10abe43ae8b3b8b22611507e3e97827438c9fdb7e155852ee","abstract_canon_sha256":"a5e02780077b5ccbb97a6b4df2ace0760742392a65b183bb7cdd2988d949f525"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:08.490938Z","signature_b64":"B6DERH0ODB5ZsKmD8i9eJityTuhLHxlh5P7xpH4R5h7fM2BRmaBWdLBT7aFcBco8MLCeV2DK1irhKeby3GwwCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8dd4723971f5a064246d99464bbc3bf82555832afaeafdfd1501fb5b7eb12ee","last_reissued_at":"2026-07-05T08:56:08.490530Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:08.490530Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ASVspoof 5: Crowdsourced Speech Data, Deepfakes, and Adversarial Attacks at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"eess.AS","authors_text":"Hector Delgado, Hemlata Tak, Hye-Jin Shim, Ivan Kukanov, Jee-weon Jung, Junichi Yamagishi, Kong Aik Lee, Massimiliano Todisco, Md Sahidullah, Nicholas Evans, Tomi Kinnunen, Xin Wang, Xuechen Liu","submitted_at":"2024-08-16T13:37:20Z","abstract_excerpt":"ASVspoof 5 is the fifth edition in a series of challenges that promote the study of speech spoofing and deepfake attacks, and the design of detection solutions. Compared to previous challenges, the ASVspoof 5 database is built from crowdsourced data collected from a vastly greater number of speakers in diverse acoustic conditions. Attacks, also crowdsourced, are generated and tested using surrogate detection models, while adversarial attacks are incorporated for the first time. New metrics support the evaluation of spoofing-robust automatic speaker verification (SASV) as well as stand-alone de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.08739","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.08739/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.08739","created_at":"2026-07-05T08:56:08.490585+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.08739v1","created_at":"2026-07-05T08:56:08.490585+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.08739","created_at":"2026-07-05T08:56:08.490585+00:00"},{"alias_kind":"pith_short_12","alias_value":"XDOUOI4XD5NA","created_at":"2026-07-05T08:56:08.490585+00:00"},{"alias_kind":"pith_short_16","alias_value":"XDOUOI4XD5NAMQSG","created_at":"2026-07-05T08:56:08.490585+00:00"},{"alias_kind":"pith_short_8","alias_value":"XDOUOI4X","created_at":"2026-07-05T08:56:08.490585+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21735","citing_title":"Bridging the Age Gap: Towards Detecting Neural Audio Codec Synthesized Elderly Speech Deepfake","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05101","citing_title":"FoeGlass: Simple In-Context Learning Is Enough for Red Teaming Audio Deepfake Detectors","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30791","citing_title":"Probing-Guided Layer Selection from Self-Supervised Speech Models for Generalizable Audio Deepfake Detection","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28064","citing_title":"I Hear, Therefore I Trust: A Socio-Technical Investigation of Humans as Synthetic Speech Detectors","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23201","citing_title":"MixFake: Benchmarking and Enhancing Audio Deepfake Detection in Diverse Real-world Mixed Audio","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2510.19414","citing_title":"EchoFake: A Replay-Aware Dataset for Practical Speech Deepfake Detection","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09007","citing_title":"Gender Fairness in Audio Deepfake Detection: Performance and Disparity Analysis","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09087","citing_title":"Towards Trustworthy Audio Deepfake Detection: A Systematic Framework for Diagnosing and Mitigating Gender Bias","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00251","citing_title":"Alethia: A Foundational Encoder for Voice Deepfakes","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08184","citing_title":"AT-ADD: All-Type Audio Deepfake Detection Challenge Evaluation Plan","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13229","citing_title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19949","citing_title":"Indic-CodecFake meets SATYAM: Towards Detecting Neural Audio Codec Synthesized Speech Deepfakes in Indic Languages","ref_index":146,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6","json":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6.json","graph_json":"https://pith.science/api/pith-number/XDOUOI4XD5NAMQSG3GKGJO6DX6/graph.json","events_json":"https://pith.science/api/pith-number/XDOUOI4XD5NAMQSG3GKGJO6DX6/events.json","paper":"https://pith.science/paper/XDOUOI4X"},"agent_actions":{"view_html":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6","download_json":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6.json","view_paper":"https://pith.science/paper/XDOUOI4X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.08739&json=true","fetch_graph":"https://pith.science/api/pith-number/XDOUOI4XD5NAMQSG3GKGJO6DX6/graph.json","fetch_events":"https://pith.science/api/pith-number/XDOUOI4XD5NAMQSG3GKGJO6DX6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6/action/storage_attestation","attest_author":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6/action/author_attestation","sign_citation":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6/action/citation_signature","submit_replication":"https://pith.science/pith/XDOUOI4XD5NAMQSG3GKGJO6DX6/action/replication_record"}},"created_at":"2026-07-05T08:56:08.490585+00:00","updated_at":"2026-07-05T08:56:08.490585+00:00"}