{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:F6QCWKIX27G6DGSTQEW3MCBLN2","short_pith_number":"pith:F6QCWKIX","schema_version":"1.0","canonical_sha256":"2fa02b2917d7cde19a53812db6082b6e8265ae9e195100c589952cd76c049d47","source":{"kind":"arxiv","id":"2003.11982","version":2},"attestation_state":"computed","paper":{"title":"In defence of metric learning for speaker recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Bong-Jin Lee, Chiheon Ham, Hee Soo Heo, Icksang Han, Jaesung Huh, Joon Son Chung, Minjae Lee, Seongkyu Mun, Soyeon Choe, Sunghwan Jung","submitted_at":"2020-03-26T15:43:10Z","abstract_excerpt":"The objective of this paper is 'open-set' speaker recognition of unseen speakers, where ideal embeddings should be able to condense information into a compact utterance-level representation that has small intra-speaker and large inter-speaker distance.\n  A popular belief in speaker recognition is that networks trained with classification objectives outperform metric learning methods. In this paper, we present an extensive evaluation of most popular loss functions for speaker recognition on the VoxCeleb dataset. We demonstrate that the vanilla triplet loss shows competitive performance compared"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.11982","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2020-03-26T15:43:10Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"116fb885cfa682c6972865169ed9b1c7c395c570bdd609dc9c17b4a858f098b1","abstract_canon_sha256":"4113b8469019384463e27fed36dbd0c6bd4c8ca55edb0b3cd3b68b1f901d227d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:49:02.496335Z","signature_b64":"Zy+0uFOwdZIoZbAhfB+mRb6Om2zVi5V1mE/A4Z0qb7vaBDGYXPoM9jAuTybnNULgP664tWKjEEqTAbP7L7qnCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fa02b2917d7cde19a53812db6082b6e8265ae9e195100c589952cd76c049d47","last_reissued_at":"2026-07-05T01:49:02.495816Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:49:02.495816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"In defence of metric learning for speaker recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Bong-Jin Lee, Chiheon Ham, Hee Soo Heo, Icksang Han, Jaesung Huh, Joon Son Chung, Minjae Lee, Seongkyu Mun, Soyeon Choe, Sunghwan Jung","submitted_at":"2020-03-26T15:43:10Z","abstract_excerpt":"The objective of this paper is 'open-set' speaker recognition of unseen speakers, where ideal embeddings should be able to condense information into a compact utterance-level representation that has small intra-speaker and large inter-speaker distance.\n  A popular belief in speaker recognition is that networks trained with classification objectives outperform metric learning methods. In this paper, we present an extensive evaluation of most popular loss functions for speaker recognition on the VoxCeleb dataset. We demonstrate that the vanilla triplet loss shows competitive performance compared"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.11982","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.11982/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.11982","created_at":"2026-07-05T01:49:02.495873+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.11982v2","created_at":"2026-07-05T01:49:02.495873+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.11982","created_at":"2026-07-05T01:49:02.495873+00:00"},{"alias_kind":"pith_short_12","alias_value":"F6QCWKIX27G6","created_at":"2026-07-05T01:49:02.495873+00:00"},{"alias_kind":"pith_short_16","alias_value":"F6QCWKIX27G6DGST","created_at":"2026-07-05T01:49:02.495873+00:00"},{"alias_kind":"pith_short_8","alias_value":"F6QCWKIX","created_at":"2026-07-05T01:49:02.495873+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23354","citing_title":"Explainable AI in Speaker Recognition -- Making Latent Representations Understandable","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22901","citing_title":"Explainable AI in Speaker Recognition -- Attention Map Visualisation and Evaluation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23354","citing_title":"Explainable AI in Speaker Recognition -- Making Latent Representations Understandable","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2","json":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2.json","graph_json":"https://pith.science/api/pith-number/F6QCWKIX27G6DGSTQEW3MCBLN2/graph.json","events_json":"https://pith.science/api/pith-number/F6QCWKIX27G6DGSTQEW3MCBLN2/events.json","paper":"https://pith.science/paper/F6QCWKIX"},"agent_actions":{"view_html":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2","download_json":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2.json","view_paper":"https://pith.science/paper/F6QCWKIX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.11982&json=true","fetch_graph":"https://pith.science/api/pith-number/F6QCWKIX27G6DGSTQEW3MCBLN2/graph.json","fetch_events":"https://pith.science/api/pith-number/F6QCWKIX27G6DGSTQEW3MCBLN2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2/action/storage_attestation","attest_author":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2/action/author_attestation","sign_citation":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2/action/citation_signature","submit_replication":"https://pith.science/pith/F6QCWKIX27G6DGSTQEW3MCBLN2/action/replication_record"}},"created_at":"2026-07-05T01:49:02.495873+00:00","updated_at":"2026-07-05T01:49:02.495873+00:00"}