{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/in-defence-of-metric-learning-for-speaker","title":"In defence of metric learning for speaker recognition","arxiv_id":"2003.11982","date":"2020-03-26","proceeding":null,"authors":["Joon Son Chung","Jaesung Huh","Seongkyu Mun","Minjae Lee","Hee Soo Heo","Soyeon Choe","Chiheon Ham","Sunghwan Jung","Bong-Jin Lee","Icksang Han"],"abstract":"The objective of this paper is 'open-set' speaker recognition of unseen speakers, where ideal embeddings should be able to condense information into a compact utterance-level representation that has small intra-speaker and large inter-speaker distance. A popular belief in speaker recognition is that networks trained with classification objectives outperform metric learning methods. In this paper, we present an extensive evaluation of most popular loss functions for speaker recognition on the VoxCeleb dataset. We demonstrate that the vanilla triplet loss shows competitive performance compared to classification-based losses, and those trained with our proposed metric learning objective outperform state-of-the-art methods.","url_abs":"http://arxiv.org/abs/2003.11982v2","url_pdf":"http://arxiv.org/pdf/2003.11982v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"links_only","authors_date_abstract":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license), from the Kaggle arXiv metadata snapshot of 2026-09-12"},"code_links":[{"paper_slug":"in-defence-of-metric-learning-for-speaker","repo_url":"https://github.com/clovaai/voxceleb_trainer","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"in-defence-of-metric-learning-for-speaker","repo_url":"https://github.com/coqui-ai/TTS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MPL-2.0"}},{"paper_slug":"in-defence-of-metric-learning-for-speaker","repo_url":"https://github.com/shkim816/decomposed_temporal_dynamic_cnn","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"in-defence-of-metric-learning-for-speaker","repo_url":"https://github.com/shkim816/temporal_dynamic_cnn","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/real-time-semantic-segmentation-on-cityscapes-1","task":"Real-Time Semantic Segmentation","dataset":"Cityscapes val","model":"SwiftNetRN-18","rank_in_archive_order":16,"of":24,"metrics":{"Frame (fps)":"39.9","mIoU":"75.5%"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2003.11982","atlas_url":"https://app.syntology.ai/?focus=2003.11982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11982"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/shkim816/decomposed_temporal_dynamic_cnn","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/clovaai/voxceleb_trainer","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/coqui-ai/TTS","reach":{"status":"ok","spdx":"MPL-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/shkim816/temporal_dynamic_cnn","reach":{"status":"ok"}}],"summary":{"ran_draft_wrong":2,"ran_violates":1},"by_repo_kind":{"official":{"samples":2,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"3a86e2f4675702d4","entry":"find_option_type","repo":"clovaai/voxceleb_trainer","repo_kind":"official","path":"trainSpeakerNet.py","file_url":"https://github.com/clovaai/voxceleb_trainer/blob/HEAD/trainSpeakerNet.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"3a86e2f4675702d4"}},{"code_sha256_prefix":"1236c5af96d325a8","entry":"is_within_directory","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"1236c5af96d325a8"}},{"code_sha256_prefix":"77c379958bd0a3ed","entry":"md5","repo":"clovaai/voxceleb_trainer","repo_kind":"official","path":"dataprep.py","file_url":"https://github.com/clovaai/voxceleb_trainer/blob/HEAD/dataprep.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"77c379958bd0a3ed"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}