{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I4I6ZEKMQTUQOKPMPOTG2K44IE","short_pith_number":"pith:I4I6ZEKM","schema_version":"1.0","canonical_sha256":"4711ec914c84e90729ec7ba66d2b9c41244374dc4392da2896f3d50fb37d98cf","source":{"kind":"arxiv","id":"2408.07414","version":1},"attestation_state":"computed","paper":{"title":"WavLM model ensemble for audio deepfake detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Adriana Stan, Dan Oneata, David Combei, Horia Cucu","submitted_at":"2024-08-14T09:43:35Z","abstract_excerpt":"Audio deepfake detection has become a pivotal task over the last couple of years, as many recent speech synthesis and voice cloning systems generate highly realistic speech samples, thus enabling their use in malicious activities. In this paper we address the issue of audio deepfake detection as it was set in the ASVspoof5 challenge. First, we benchmark ten types of pretrained representations and show that the self-supervised representations stemming from the wav2vec2 and wavLM families perform best. Of the two, wavLM is better when restricting the pretraining data to LibriSpeech, as required "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.07414","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-08-14T09:43:35Z","cross_cats_sorted":[],"title_canon_sha256":"d8c5decef4c8df0465ab3ee9dac6414d367c4f6f6d05f935f91fddc5fa5fd09c","abstract_canon_sha256":"b69242ed3b854c5880758c0307b28de1e2c90a71b65db69bd3bfcc484aa1727d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:24.987279Z","signature_b64":"DImM+2zXNRrxBfXTgvZmbRoCe0kpAgbLQPYIw94KauAQAtj/S1Ohelmxoqx/QqUm/APC4+Np8FM1lBIc+k4sAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4711ec914c84e90729ec7ba66d2b9c41244374dc4392da2896f3d50fb37d98cf","last_reissued_at":"2026-07-05T08:55:24.986782Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:24.986782Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WavLM model ensemble for audio deepfake detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Adriana Stan, Dan Oneata, David Combei, Horia Cucu","submitted_at":"2024-08-14T09:43:35Z","abstract_excerpt":"Audio deepfake detection has become a pivotal task over the last couple of years, as many recent speech synthesis and voice cloning systems generate highly realistic speech samples, thus enabling their use in malicious activities. In this paper we address the issue of audio deepfake detection as it was set in the ASVspoof5 challenge. First, we benchmark ten types of pretrained representations and show that the self-supervised representations stemming from the wav2vec2 and wavLM families perform best. Of the two, wavLM is better when restricting the pretraining data to LibriSpeech, as required "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.07414","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.07414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.07414","created_at":"2026-07-05T08:55:24.986851+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.07414v1","created_at":"2026-07-05T08:55:24.986851+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.07414","created_at":"2026-07-05T08:55:24.986851+00:00"},{"alias_kind":"pith_short_12","alias_value":"I4I6ZEKMQTUQ","created_at":"2026-07-05T08:55:24.986851+00:00"},{"alias_kind":"pith_short_16","alias_value":"I4I6ZEKMQTUQOKPM","created_at":"2026-07-05T08:55:24.986851+00:00"},{"alias_kind":"pith_short_8","alias_value":"I4I6ZEKM","created_at":"2026-07-05T08:55:24.986851+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10758","citing_title":"Anchoring the Unknown: Open-Set Model Attribution via Proxy-Anchor Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27557","citing_title":"A General Model for Deepfake Speech Detection: Diverse Bonafide Resources or Diverse AI-Based Generators","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE","json":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE.json","graph_json":"https://pith.science/api/pith-number/I4I6ZEKMQTUQOKPMPOTG2K44IE/graph.json","events_json":"https://pith.science/api/pith-number/I4I6ZEKMQTUQOKPMPOTG2K44IE/events.json","paper":"https://pith.science/paper/I4I6ZEKM"},"agent_actions":{"view_html":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE","download_json":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE.json","view_paper":"https://pith.science/paper/I4I6ZEKM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.07414&json=true","fetch_graph":"https://pith.science/api/pith-number/I4I6ZEKMQTUQOKPMPOTG2K44IE/graph.json","fetch_events":"https://pith.science/api/pith-number/I4I6ZEKMQTUQOKPMPOTG2K44IE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE/action/storage_attestation","attest_author":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE/action/author_attestation","sign_citation":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE/action/citation_signature","submit_replication":"https://pith.science/pith/I4I6ZEKMQTUQOKPMPOTG2K44IE/action/replication_record"}},"created_at":"2026-07-05T08:55:24.986851+00:00","updated_at":"2026-07-05T08:55:24.986851+00:00"}