{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YDFA5GZCCXUKVZFGJJYSG4QJ3K","short_pith_number":"pith:YDFA5GZC","schema_version":"1.0","canonical_sha256":"c0ca0e9b2215e8aae4a64a71237209daa3b5ed04738e04c2a9bbfc845691b605","source":{"kind":"arxiv","id":"2401.08342","version":1},"attestation_state":"computed","paper":{"title":"ECAPA2: A Hybrid Neural Network Architecture and Training Strategy for Robust Speaker Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Jenthe Thienpondt, Kris Demuynck","submitted_at":"2024-01-16T13:17:39Z","abstract_excerpt":"In this paper, we present ECAPA2, a novel hybrid neural network architecture and training strategy to produce robust speaker embeddings. Most speaker verification models are based on either the 1D- or 2D-convolutional operation, often manifested as Time Delay Neural Networks or ResNets, respectively. Hybrid models are relatively unexplored without an intuitive explanation what constitutes best practices in regard to its architectural choices. We motivate the proposed ECAPA2 model in this paper with an analysis of current speaker verification architectures. In addition, we propose a training st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.08342","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2024-01-16T13:17:39Z","cross_cats_sorted":[],"title_canon_sha256":"2adef80fd69eb3d04415c4e32dd715a3b28684e9af1518bf5866c53b0351d05d","abstract_canon_sha256":"84b11e731975f6e9ad0e3a9d540494d7e14424493030048b3625f17e61ad3db8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:34:10.756163Z","signature_b64":"48vTcwiH2Tsa2L3bK7G14TLHoybk7XIxNwmMmGnVWd0LC1/I2GH4Ae2nknwyMKfnMU8Ih8xL4QcMvty6NZywBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0ca0e9b2215e8aae4a64a71237209daa3b5ed04738e04c2a9bbfc845691b605","last_reissued_at":"2026-07-05T07:34:10.755718Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:34:10.755718Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ECAPA2: A Hybrid Neural Network Architecture and Training Strategy for Robust Speaker Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Jenthe Thienpondt, Kris Demuynck","submitted_at":"2024-01-16T13:17:39Z","abstract_excerpt":"In this paper, we present ECAPA2, a novel hybrid neural network architecture and training strategy to produce robust speaker embeddings. Most speaker verification models are based on either the 1D- or 2D-convolutional operation, often manifested as Time Delay Neural Networks or ResNets, respectively. Hybrid models are relatively unexplored without an intuitive explanation what constitutes best practices in regard to its architectural choices. We motivate the proposed ECAPA2 model in this paper with an analysis of current speaker verification architectures. In addition, we propose a training st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.08342","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.08342/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.08342","created_at":"2026-07-05T07:34:10.755777+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.08342v1","created_at":"2026-07-05T07:34:10.755777+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.08342","created_at":"2026-07-05T07:34:10.755777+00:00"},{"alias_kind":"pith_short_12","alias_value":"YDFA5GZCCXUK","created_at":"2026-07-05T07:34:10.755777+00:00"},{"alias_kind":"pith_short_16","alias_value":"YDFA5GZCCXUKVZFG","created_at":"2026-07-05T07:34:10.755777+00:00"},{"alias_kind":"pith_short_8","alias_value":"YDFA5GZC","created_at":"2026-07-05T07:34:10.755777+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.05835","citing_title":"NanoCodec: Towards High-Quality Ultra Fast Speech LLM Inference","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K","json":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K.json","graph_json":"https://pith.science/api/pith-number/YDFA5GZCCXUKVZFGJJYSG4QJ3K/graph.json","events_json":"https://pith.science/api/pith-number/YDFA5GZCCXUKVZFGJJYSG4QJ3K/events.json","paper":"https://pith.science/paper/YDFA5GZC"},"agent_actions":{"view_html":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K","download_json":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K.json","view_paper":"https://pith.science/paper/YDFA5GZC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.08342&json=true","fetch_graph":"https://pith.science/api/pith-number/YDFA5GZCCXUKVZFGJJYSG4QJ3K/graph.json","fetch_events":"https://pith.science/api/pith-number/YDFA5GZCCXUKVZFGJJYSG4QJ3K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K/action/storage_attestation","attest_author":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K/action/author_attestation","sign_citation":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K/action/citation_signature","submit_replication":"https://pith.science/pith/YDFA5GZCCXUKVZFGJJYSG4QJ3K/action/replication_record"}},"created_at":"2026-07-05T07:34:10.755777+00:00","updated_at":"2026-07-05T07:34:10.755777+00:00"}