{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/voice-conversion-with-just-nearest-neighbors","title":"Voice Conversion With Just Nearest Neighbors","arxiv_id":"2305.18975","date":"2023-05-30","proceeding":null,"authors":["Matthew Baas","Benjamin van Niekerk","Herman Kamper"],"abstract":"Any-to-any voice conversion aims to transform source speech into a target voice with just a few examples of the target speaker as a reference. Recent methods produce convincing conversions, but at the cost of increased complexity -- making results difficult to reproduce and build on. Instead, we keep it simple. We propose k-nearest neighbors voice conversion (kNN-VC): a straightforward yet effective method for any-to-any conversion. First, we extract self-supervised representations of the source and reference speech. To convert to the target speaker, we replace each frame of the source representation with its nearest neighbor in the reference. Finally, a pretrained vocoder synthesizes audio from the converted representation. Objective and subjective evaluations show that kNN-VC improves speaker similarity with similar intelligibility scores to existing methods. Code, samples, trained models: https://bshall.github.io/knn-vc","url_abs":"https://arxiv.org/abs/2305.18975v1","url_pdf":"https://arxiv.org/pdf/2305.18975v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"voice-conversion-with-just-nearest-neighbors","repo_url":"https://github.com/bshall/knn-vc","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}}],"tasks":[{"task_slug":"voice-conversion","task_name":"Voice Conversion"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/voice-conversion-on-librispeech-test-clean","task":"Voice Conversion","dataset":"LibriSpeech test-clean","model":"kNN-VC (prematched HiFiGAN)","rank_in_archive_order":1,"of":1,"metrics":{"Character Error Rate (CER)":"2.96","Equal Error Rate":"37.15","Word Error Rate (WER)":"7.36"},"uses_additional_data":true}],"syntology":{"syntology_url":"https://syntology.ai/paper/2305.18975","atlas_url":"https://app.syntology.ai/?focus=2305.18975","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}