{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/voice-conversion/papers/2","list_of":"/task/voice-conversion","task":"Voice Conversion","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":6,"rows_per_page":100,"rows":[101,200],"of":520,"counts":{"archive_papers_tagged":520,"with_a_code_link":175,"where_syntology_ran_a_sample":41,"not_listed_spam_title":0,"listed":520,"listed_where_code_ran":41,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":32,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":32,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/voice-conversion","prev":"/task/voice-conversion","next":"/task/voice-conversion/papers/3","papers":[{"url":"/paper/speaking-style-conversion-with-discrete-self","slug":"speaking-style-conversion-with-discrete-self","title":"Speaking Style Conversion in the Waveform Domain Using Discrete Self-Supervised Units","date":"2022-12-19","arxiv_id":"2212.09730","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speaking-style-conversion-with-discrete-self#ran","syntology_url":"https://syntology.ai/paper/2212.09730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09730"}},"official":{"repos":["gallilmaimon/DISSC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hiding-speaker-s-sex-in-speech-using-zero","slug":"hiding-speaker-s-sex-in-speech-using-zero","title":"Hiding speaker's sex in speech using zero-evidence speaker representation in an analysis/synthesis pipeline","date":"2022-11-29","arxiv_id":"2211.16065","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-one-shot-prosody-and-speaker","slug":"a-unified-one-shot-prosody-and-speaker","title":"A unified one-shot prosody and speaker conversion system with self-supervised discrete speech units","date":"2022-11-12","arxiv_id":"2211.06535","repositories_listed":1,"syntology":null},{"url":"/paper/freevc-towards-high-quality-text-free-one","slug":"freevc-towards-high-quality-text-free-one","title":"FreeVC: Towards High-Quality Text-Free One-Shot Voice Conversion","date":"2022-10-27","arxiv_id":"2210.15418","repositories_listed":1,"syntology":null},{"url":"/paper/gan-you-hear-me-reclaiming-unconditional","slug":"gan-you-hear-me-reclaiming-unconditional","title":"GAN You Hear Me? Reclaiming Unconditional Speech Synthesis from Diffusion Models","date":"2022-10-11","arxiv_id":"2210.05271","repositories_listed":1,"syntology":null},{"url":"/paper/voice-spoofing-countermeasures-taxonomy-state","slug":"voice-spoofing-countermeasures-taxonomy-state","title":"Voice Spoofing Countermeasures: Taxonomy, State-of-the-art, experimental analysis of generalizability, open challenges, and the way forward","date":"2022-10-02","arxiv_id":"2210.00417","repositories_listed":1,"syntology":null},{"url":"/paper/controlvc-zero-shot-voice-conversion-with","slug":"controlvc-zero-shot-voice-conversion-with","title":"ControlVC: Zero-Shot Voice Conversion with Time-Varying Controls on Pitch and Speed","date":"2022-09-23","arxiv_id":"2209.11866","repositories_listed":1,"syntology":null},{"url":"/paper/deid-vc-speaker-de-identification-via-zero","slug":"deid-vc-speaker-de-identification-via-zero","title":"DeID-VC: Speaker De-identification via Zero-shot Pseudo Voice Conversion","date":"2022-09-09","arxiv_id":"2209.04530","repositories_listed":1,"syntology":null},{"url":"/paper/speech-representation-disentanglement-with","slug":"speech-representation-disentanglement-with","title":"Speech Representation Disentanglement with Adversarial Mutual Information Learning for One-shot Voice Conversion","date":"2022-08-18","arxiv_id":"2208.08757","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-self-supervised-speech","slug":"a-comparative-study-of-self-supervised-speech","title":"A Comparative Study of Self-supervised Speech Representation Based Voice Conversion","date":"2022-07-10","arxiv_id":"2207.04356","repositories_listed":1,"syntology":null},{"url":"/paper/accurate-emotion-strength-assessment-for-seen","slug":"accurate-emotion-strength-assessment-for-seen","title":"Accurate Emotion Strength Assessment for Seen and Unseen Speech Based on Data-Driven Deep Learning","date":"2022-06-15","arxiv_id":"2206.07229","repositories_listed":1,"syntology":null},{"url":"/paper/speak-like-a-dog-human-to-non-human-creature","slug":"speak-like-a-dog-human-to-non-human-creature","title":"Speak Like a Dog: Human to Non-human creature Voice Conversion","date":"2022-06-09","arxiv_id":"2206.04780","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-zero-shot-voice-style-transfer","slug":"end-to-end-zero-shot-voice-style-transfer","title":"End-to-End Zero-Shot Voice Conversion with Location-Variable Convolutions","date":"2022-05-19","arxiv_id":"2205.09784","repositories_listed":1,"syntology":null},{"url":"/paper/towards-improved-zero-shot-voice-conversion","slug":"towards-improved-zero-shot-voice-conversion","title":"Towards Improved Zero-shot Voice Conversion with Conditional DSVAE","date":"2022-05-11","arxiv_id":"2205.05227","repositories_listed":1,"syntology":null},{"url":"/paper/read-the-room-adapting-a-robot-s-voice-to","slug":"read-the-room-adapting-a-robot-s-voice-to","title":"Read the Room: Adapting a Robot's Voice to Ambient and Social Contexts","date":"2022-05-10","arxiv_id":"2205.04952","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-learning-improves-synthetic-speech","slug":"multi-task-learning-improves-synthetic-speech","title":"Multi-task learning improves synthetic speech detection","date":"2022-04-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/universal-adaptor-converting-mel-spectrograms","slug":"universal-adaptor-converting-mel-spectrograms","title":"Universal Adaptor: Converting Mel-Spectrograms Between Different Configurations for Speech Synthesis","date":"2022-04-01","arxiv_id":"2204.00170","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-non-autoregressive-gan-voice","slug":"efficient-non-autoregressive-gan-voice","title":"Efficient Non-Autoregressive GAN Voice Conversion using VQWav2vec Features and Dynamic Convolution","date":"2022-03-31","arxiv_id":"2203.17172","repositories_listed":1,"syntology":null},{"url":"/paper/hifi-vc-high-quality-asr-based-voice","slug":"hifi-vc-high-quality-asr-based-voice","title":"HiFi-VC: High Quality ASR-Based Voice Conversion","date":"2022-03-31","arxiv_id":"2203.16937","repositories_listed":1,"syntology":null},{"url":"/paper/robust-disentangled-variational-speech","slug":"robust-disentangled-variational-speech","title":"Robust Disentangled Variational Speech Representation Learning for Zero-shot Voice Conversion","date":"2022-03-30","arxiv_id":"2203.16705","repositories_listed":1,"syntology":null},{"url":"/paper/a-single-speaker-is-almost-all-you-need-for","slug":"a-single-speaker-is-almost-all-you-need-for","title":"ASR data augmentation in low-resource settings using cross-lingual multi-speaker TTS and cross-lingual voice conversion","date":"2022-03-29","arxiv_id":"2204.00618","repositories_listed":1,"syntology":null},{"url":"/paper/speechsplit-2-0-unsupervised-speech","slug":"speechsplit-2-0-unsupervised-speech","title":"SpeechSplit 2.0: Unsupervised speech disentanglement for voice conversion Without tuning autoencoder Bottlenecks","date":"2022-03-26","arxiv_id":"2203.14156","repositories_listed":1,"syntology":null},{"url":"/paper/towards-privacy-preserving-speech","slug":"towards-privacy-preserving-speech","title":"A Speech Representation Anonymization Framework via Selective Noise Perturbation","date":"2022-03-26","arxiv_id":"2203.14171","repositories_listed":1,"syntology":null},{"url":"/paper/a-practical-guide-to-logical-access-voice","slug":"a-practical-guide-to-logical-access-voice","title":"A Practical Guide to Logical Access Voice Presentation Attack Detection","date":"2022-01-10","arxiv_id":"2201.03321","repositories_listed":1,"syntology":null},{"url":"/paper/cycletransgan-evc-a-cyclegan-based-emotional","slug":"cycletransgan-evc-a-cyclegan-based-emotional","title":"CycleTransGAN-EVC: A CycleGAN-based Emotional Voice Conversion Model with Transformer","date":"2021-11-30","arxiv_id":"2111.15159","repositories_listed":1,"syntology":null},{"url":"/paper/sig-vc-a-speaker-information-guided-zero-shot","slug":"sig-vc-a-speaker-information-guided-zero-shot","title":"SIG-VC: A Speaker Information Guided Zero-shot Voice Conversion System for Both Human Beings and Machines","date":"2021-11-06","arxiv_id":"2111.03811","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-and-interpretable-singing-voice","slug":"controllable-and-interpretable-singing-voice","title":"Controllable and Interpretable Singing Voice Decomposition via Assem-VC","date":"2021-10-25","arxiv_id":"2110.12676","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/controllable-and-interpretable-singing-voice#ran","syntology_url":"https://syntology.ai/paper/2110.12676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.12676"}},"official":{"repos":["mindslab-ai/assem-vc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fmfcc-a-a-challenging-mandarin-dataset-for","slug":"fmfcc-a-a-challenging-mandarin-dataset-for","title":"FMFCC-A: A Challenging Mandarin Dataset for Synthetic Speech Detection","date":"2021-10-18","arxiv_id":"2110.09441","repositories_listed":1,"syntology":null},{"url":"/paper/ldnet-unified-listener-dependent-modeling-in","slug":"ldnet-unified-listener-dependent-modeling-in","title":"LDNet: Unified Listener Dependent Modeling in MOS Prediction for Synthetic Speech","date":"2021-10-18","arxiv_id":"2110.09103","repositories_listed":1,"syntology":null},{"url":"/paper/toward-degradation-robust-voice-conversion","slug":"toward-degradation-robust-voice-conversion","title":"Toward Degradation-Robust Voice Conversion","date":"2021-10-14","arxiv_id":"2110.07537","repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-speaker-independent-emotions-for","slug":"decoupling-speaker-independent-emotions-for","title":"Decoupling Speaker-Independent Emotions for Voice Conversion Via Source-Filter Networks","date":"2021-10-04","arxiv_id":"2110.01164","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-text-to-speech-for-text-based","slug":"zero-shot-text-to-speech-for-text-based","title":"Zero-Shot Text-to-Speech for Text-Based Insertion in Audio Narration","date":"2021-09-12","arxiv_id":"2109.05426","repositories_listed":1,"syntology":null},{"url":"/paper/svsnet-an-end-to-end-speaker-voice-similarity","slug":"svsnet-an-end-to-end-speaker-voice-similarity","title":"SVSNet: An End-to-end Speaker Voice Similarity Assessment Model","date":"2021-07-20","arxiv_id":"2107.09392","repositories_listed":1,"syntology":null},{"url":"/paper/an-improved-stargan-for-emotional-voice","slug":"an-improved-stargan-for-emotional-voice","title":"An Improved StarGAN for Emotional Voice Conversion: Enhancing Voice Quality and Data Augmentation","date":"2021-07-18","arxiv_id":"2107.08361","repositories_listed":1,"syntology":null},{"url":"/paper/vqmivc-vector-quantization-and-mutual","slug":"vqmivc-vector-quantization-and-mutual","title":"VQMIVC: Vector Quantization and Mutual Information-Based Unsupervised Speech Representation Disentanglement for One-shot Voice Conversion","date":"2021-06-18","arxiv_id":"2106.10132","repositories_listed":1,"syntology":null},{"url":"/paper/voicy-zero-shot-non-parallel-voice-conversion","slug":"voicy-zero-shot-non-parallel-voice-conversion","title":"Voicy: Zero-Shot Non-Parallel Voice Conversion in Noisy Reverberant Environments","date":"2021-06-16","arxiv_id":"2106.08873","repositories_listed":1,"syntology":null},{"url":"/paper/nvc-net-end-to-end-adversarial-voice","slug":"nvc-net-end-to-end-adversarial-voice","title":"NVC-Net: End-to-End Adversarial Voice Conversion","date":"2021-06-02","arxiv_id":"2106.00992","repositories_listed":1,"syntology":null},{"url":"/paper/emotional-voice-conversion-theory-databases","slug":"emotional-voice-conversion-theory-databases","title":"Emotional Voice Conversion: Theory, Databases and ESD","date":"2021-05-31","arxiv_id":"2105.14762","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-based-assessment-of-synthetic","slug":"deep-learning-based-assessment-of-synthetic","title":"Deep Learning Based Assessment of Synthetic Speech Naturalness","date":"2021-04-23","arxiv_id":"2104.11673","repositories_listed":1,"syntology":null},{"url":"/paper/building-bilingual-and-code-switched-voice","slug":"building-bilingual-and-code-switched-voice","title":"Building Bilingual and Code-Switched Voice Conversion with Limited Training Data Using Embedding Consistency Loss","date":"2021-04-22","arxiv_id":"2104.10832","repositories_listed":1,"syntology":null},{"url":"/paper/assem-vc-realistic-voice-conversion-by","slug":"assem-vc-realistic-voice-conversion-by","title":"Assem-VC: Realistic Voice Conversion by Assembling Modern Speech Synthesis Techniques","date":"2021-04-02","arxiv_id":"2104.00931","repositories_listed":1,"syntology":null},{"url":"/paper/improving-zero-shot-voice-style-transfer-via-1","slug":"improving-zero-shot-voice-style-transfer-via-1","title":"Improving Zero-shot Voice Style Transfer via Disentangled Representation Learning","date":"2021-03-17","arxiv_id":"2103.09420","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-zero-shot-voice-style-transfer-via-1#ran","syntology_url":"https://syntology.ai/paper/2103.09420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.09420"}},"official":null}},{"url":"/paper/investigating-on-incorporating-pretrained-and","slug":"investigating-on-incorporating-pretrained-and","title":"Investigating on Incorporating Pretrained and Learnable Speaker Representations for Multi-Speaker Multi-Style Text-to-Speech","date":"2021-03-06","arxiv_id":"2103.04088","repositories_listed":1,"syntology":null},{"url":"/paper/crank-an-open-source-software-for-nonparallel","slug":"crank-an-open-source-software-for-nonparallel","title":"crank: An Open-Source Software for Nonparallel Voice Conversion Based on Vector-Quantized Variational Autoencoder","date":"2021-03-04","arxiv_id":"2103.02858","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-disentanglement-of-speaker","slug":"adversarial-disentanglement-of-speaker","title":"Adversarial Disentanglement of Speaker Representation for Attribute-Driven Privacy Preservation","date":"2020-12-08","arxiv_id":"2012.04454","repositories_listed":1,"syntology":null},{"url":"/paper/phonetic-posteriorgrams-based-many-to-many","slug":"phonetic-posteriorgrams-based-many-to-many","title":"Phonetic Posteriorgrams based Many-to-Many Singing Voice Conversion via Adversarial Training","date":"2020-12-03","arxiv_id":"2012.01837","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/phonetic-posteriorgrams-based-many-to-many#ran","syntology_url":"https://syntology.ai/paper/2012.01837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.01837"}},"official":{"repos":["hhguo/EA-SVC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voice-conversion-using-speech-to-speech-neuro","slug":"voice-conversion-using-speech-to-speech-neuro","title":"Voice Conversion Using Speech-to-Speech Neuro-Style Transfer","date":"2020-10-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-natural-bilingual-and-code-switched","slug":"towards-natural-bilingual-and-code-switched","title":"Towards Natural Bilingual and Code-Switched Speech Synthesis Based on Mix of Monolingual Recordings and Cross-Lingual Voice Conversion","date":"2020-10-16","arxiv_id":"2010.08136","repositories_listed":1,"syntology":null},{"url":"/paper/baseline-system-of-voice-conversion-challenge","slug":"baseline-system-of-voice-conversion-challenge","title":"Baseline System of Voice Conversion Challenge 2020 with Cyclic Variational Autoencoder and Parallel WaveGAN","date":"2020-10-09","arxiv_id":"2010.04429","repositories_listed":1,"syntology":null},{"url":"/paper/transfer-learning-from-monolingual-asr-to","slug":"transfer-learning-from-monolingual-asr-to","title":"Transfer Learning from Monolingual ASR to Transcription-free Cross-lingual Voice Conversion","date":"2020-09-30","arxiv_id":"2009.14668","repositories_listed":1,"syntology":null},{"url":"/paper/any-to-many-voice-conversion-with-location","slug":"any-to-many-voice-conversion-with-location","title":"Any-to-Many Voice Conversion with Location-Relative Sequence-to-Sequence Modeling","date":"2020-09-06","arxiv_id":"2009.02725","repositories_listed":1,"syntology":null},{"url":"/paper/non-parallel-voice-conversion-with-augmented","slug":"non-parallel-voice-conversion-with-augmented","title":"Nonparallel Voice Conversion with Augmented Classifier Star Generative Adversarial Networks","date":"2020-08-27","arxiv_id":"2008.12604","repositories_listed":1,"syntology":null},{"url":"/paper/cinc-gan-for-effective-f0-prediction-for","slug":"cinc-gan-for-effective-f0-prediction-for","title":"CinC-GAN for Effective F0 prediction for Whisper-to-Normal Speech Conversion","date":"2020-08-18","arxiv_id":"2008.07788","repositories_listed":1,"syntology":null},{"url":"/paper/vqvc-one-shot-voice-conversion-by-vector","slug":"vqvc-one-shot-voice-conversion-by-vector","title":"VQVC+: One-Shot Voice Conversion by Vector Quantization and U-Net architecture","date":"2020-06-07","arxiv_id":"2006.04154","repositories_listed":1,"syntology":null},{"url":"/paper/generative-adversarial-training-data","slug":"generative-adversarial-training-data","title":"Generative Adversarial Training Data Adaptation for Very Low-resource Automatic Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09256","repositories_listed":1,"syntology":null},{"url":"/paper/defending-your-voice-adversarial-attack-on","slug":"defending-your-voice-adversarial-attack-on","title":"Defending Your Voice: Adversarial Attack on Voice Conversion","date":"2020-05-18","arxiv_id":"2005.08781","repositories_listed":1,"syntology":null},{"url":"/paper/robust-training-of-vector-quantized","slug":"robust-training-of-vector-quantized","title":"Robust Training of Vector Quantized Bottleneck Models","date":"2020-05-18","arxiv_id":"2005.08520","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-training-of-vector-quantized#ran","syntology_url":"https://syntology.ai/paper/2005.08520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08520"}},"official":{"repos":["distsup/DistSup"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/converting-anyone-s-emotion-towards-speaker","slug":"converting-anyone-s-emotion-towards-speaker","title":"Converting Anyone's Emotion: Towards Speaker-Independent Emotional Voice Conversion","date":"2020-05-13","arxiv_id":"2005.07025","repositories_listed":1,"syntology":null},{"url":"/paper/f0-consistent-many-to-many-non-parallel-voice","slug":"f0-consistent-many-to-many-non-parallel-voice","title":"F0-consistent many-to-many non-parallel voice conversion via conditional autoencoder","date":"2020-04-15","arxiv_id":"2004.07370","repositories_listed":1,"syntology":null},{"url":"/paper/vocoder-free-end-to-end-voice-conversion-with","slug":"vocoder-free-end-to-end-voice-conversion-with","title":"Vocoder-free End-to-End Voice Conversion with Transformer Network","date":"2020-02-05","arxiv_id":"2002.03808","repositories_listed":1,"syntology":null},{"url":"/paper/transforming-spectrum-and-prosody-for","slug":"transforming-spectrum-and-prosody-for","title":"Transforming Spectrum and Prosody for Emotional Voice Conversion with Non-Parallel Training Data","date":"2020-02-01","arxiv_id":"2002.00198","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-representation-disentanglement","slug":"unsupervised-representation-disentanglement","title":"Unsupervised Representation Disentanglement using Cross Domain Features and Adversarial Learning in Variational Autoencoder based Voice Conversion","date":"2020-01-22","arxiv_id":"2001.07849","repositories_listed":1,"syntology":null},{"url":"/paper/emotional-voice-conversion-using-multitask","slug":"emotional-voice-conversion-using-multitask","title":"Emotional Voice Conversion using Multitask Learning with Text-to-speech","date":"2019-11-11","arxiv_id":"1911.06149","repositories_listed":1,"syntology":null},{"url":"/paper/adagan-adaptive-gan-for-many-to-many-non","slug":"adagan-adaptive-gan-for-many-to-many-non","title":"AdaGAN: Adaptive GAN for Many-to-Many Non-Parallel Voice Conversion","date":"2019-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/emotionless-privacy-preserving-speech","slug":"emotionless-privacy-preserving-speech","title":"Emotionless: Privacy-Preserving Speech Analysis for Voice Assistants","date":"2019-08-09","arxiv_id":"1908.03632","repositories_listed":1,"syntology":null},{"url":"/paper/deep-residual-neural-networks-for-audio","slug":"deep-residual-neural-networks-for-audio","title":"Deep Residual Neural Networks for Audio Spoofing Detection","date":"2019-06-30","arxiv_id":"1907.00501","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-end-to-end-learning-of-discrete","slug":"unsupervised-end-to-end-learning-of-discrete","title":"Unsupervised End-to-End Learning of Discrete Linguistic Units for Voice Conversion","date":"2019-05-28","arxiv_id":"1905.11563","repositories_listed":1,"syntology":null},{"url":"/paper/investigation-of-f0-conditioning-and-fully","slug":"investigation-of-f0-conditioning-and-fully","title":"Investigation of F0 conditioning and Fully Convolutional Networks in Variational Autoencoder based Voice Conversion","date":"2019-05-02","arxiv_id":"1905.00615","repositories_listed":1,"syntology":null},{"url":"/paper/spoof-detection-using-x-vector-and-feature","slug":"spoof-detection-using-x-vector-and-feature","title":"Spoof detection using time-delay shallow neural network and feature switching","date":"2019-04-16","arxiv_id":"1904.07453","repositories_listed":1,"syntology":null},{"url":"/paper/stc-antispoofing-systems-for-the-asvspoof2019","slug":"stc-antispoofing-systems-for-the-asvspoof2019","title":"STC Antispoofing Systems for the ASVspoof2019 Challenge","date":"2019-04-11","arxiv_id":"1904.05576","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stc-antispoofing-systems-for-the-asvspoof2019#ran","syntology_url":"https://syntology.ai/paper/1904.05576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.05576"}},"official":null}},{"url":"/paper/assert-anti-spoofing-with-squeeze-excitation","slug":"assert-anti-spoofing-with-squeeze-excitation","title":"ASSERT: Anti-Spoofing with Squeeze-Excitation and Residual neTworks","date":"2019-04-01","arxiv_id":"1904.01120","repositories_listed":1,"syntology":null},{"url":"/paper/spectrogram-channels-u-net-a-source","slug":"spectrogram-channels-u-net-a-source","title":"Spectrogram-channels u-net: a source separation model viewing each channel as the spectrogram of each source","date":"2018-10-26","arxiv_id":"1810.11520","repositories_listed":1,"syntology":null},{"url":"/paper/voice-conversion-based-on-cross-domain","slug":"voice-conversion-based-on-cross-domain","title":"Voice Conversion Based on Cross-Domain Features Using Variational Auto Encoders","date":"2018-08-29","arxiv_id":"1808.09634","repositories_listed":1,"syntology":null},{"url":"/paper/voice-conversion-from-unaligned-corpora-using","slug":"voice-conversion-from-unaligned-corpora-using","title":"Voice Conversion from Unaligned Corpora using Variational Autoencoding Wasserstein Generative Adversarial Networks","date":"2017-04-04","arxiv_id":"1704.00849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voice-conversion-from-unaligned-corpora-using#ran","syntology_url":"https://syntology.ai/paper/1704.00849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1704.00849"}},"official":null}},{"url":"/paper/voice-conversion-using-convolutional-neural","slug":"voice-conversion-using-convolutional-neural","title":"Voice Conversion using Convolutional Neural Networks","date":"2016-10-27","arxiv_id":"1610.08927","repositories_listed":1,"syntology":null},{"url":null,"slug":"2506-10289","title":"RT-VC: Real-Time Zero-Shot Voice Conversion with Speech Articulatory Coding","date":"2025-06-12","arxiv_id":"2506.10289","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-vada-a-confidence-oriented-voice","title":"CO-VADA: A Confidence-Oriented Voice Augmentation Debiasing Approach for Fair Speech Emotion Recognition","date":"2025-06-06","arxiv_id":"2506.06071","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-better-disentanglement-in-non","title":"Towards Better Disentanglement in Non-Autoregressive Zero-Shot Expressive Voice Conversion","date":"2025-06-04","arxiv_id":"2506.04013","repositories_listed":0,"syntology":null},{"url":null,"slug":"starvc-a-unified-auto-regressive-framework","title":"StarVC: A Unified Auto-Regressive Framework for Joint Text and Speech Generation in Voice Conversion","date":"2025-06-03","arxiv_id":"2506.02414","repositories_listed":0,"syntology":null},{"url":null,"slug":"linearvc-linear-transformations-of-self","title":"LinearVC: Linear transformations of self-supervised features through the lens of voice conversion","date":"2025-06-02","arxiv_id":"2506.01510","repositories_listed":0,"syntology":null},{"url":null,"slug":"salf-mos-speaker-agnostic-latent-features","title":"SALF-MOS: Speaker Agnostic Latent Features Downsampled for MOS Prediction","date":"2025-06-02","arxiv_id":"2506.02082","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudovc-improving-one-shot-voice-conversion","title":"PseudoVC: Improving One-shot Voice Conversion with Pseudo Paired Data","date":"2025-06-01","arxiv_id":"2506.01039","repositories_listed":0,"syntology":null},{"url":null,"slug":"rhythm-controllable-and-efficient-zero-shot","title":"Rhythm Controllable and Efficient Zero-Shot Voice Conversion via Shortcut Flow Matching","date":"2025-06-01","arxiv_id":"2506.01014","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-perception-based-l2-speech-intelligibility","title":"A Perception-Based L2 Speech Intelligibility Indicator: Leveraging a Rater's Shadowing and Sequence-to-sequence Voice Conversion","date":"2025-05-30","arxiv_id":"2505.24304","repositories_listed":0,"syntology":null},{"url":null,"slug":"discl-vc-disentangled-discrete-tokens-and-in","title":"Discl-VC: Disentangled Discrete Tokens and In-Context Learning for Controllable Zero-Shot Voice Conversion","date":"2025-05-30","arxiv_id":"2505.24291","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-improves-cross-domain","title":"Voice Conversion Improves Cross-Domain Robustness for Spoken Arabic Dialect Identification","date":"2025-05-30","arxiv_id":"2505.24713","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-humans-growl-and-birds-speak-high","title":"When Humans Growl and Birds Speak: High-Fidelity Voice Conversion from Human to Animal and Designed Sounds","date":"2025-05-30","arxiv_id":"2505.24336","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptevc-controllable-emotional-voice","title":"PromptEVC: Controllable Emotional Voice Conversion with Natural Language Prompts","date":"2025-05-27","arxiv_id":"2505.20678","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewind-speech-time-reversal-for-enhancing","title":"REWIND: Speech Time Reversal for Enhancing Speaker Representations in Diffusion-based Voice Conversion","date":"2025-05-27","arxiv_id":"2505.20756","repositories_listed":0,"syntology":null},{"url":null,"slug":"vibe-svc-vibrato-extraction-with-high","title":"VibE-SVC: Vibrato Extraction with High-frequency F0 Contour for Singing Voice Conversion","date":"2025-05-27","arxiv_id":"2505.20794","repositories_listed":0,"syntology":null},{"url":"/paper/arvoice-a-multi-speaker-dataset-for-arabic","slug":"arvoice-a-multi-speaker-dataset-for-arabic","title":"ArVoice: A Multi-Speaker Dataset for Arabic Speech Synthesis","date":"2025-05-26","arxiv_id":"2505.20506","repositories_listed":0,"syntology":null},{"url":null,"slug":"eta-wavlm-efficient-speaker-identity-removal","title":"Eta-WavLM: Efficient Speaker Identity Removal in Self-Supervised Speech Representations Using a Simple Linear Equation","date":"2025-05-25","arxiv_id":"2505.19273","repositories_listed":0,"syntology":null},{"url":null,"slug":"ez-vc-easy-zero-shot-any-to-any-voice","title":"EZ-VC: Easy Zero-shot Any-to-Any Voice Conversion","date":"2025-05-22","arxiv_id":"2505.16691","repositories_listed":0,"syntology":null},{"url":null,"slug":"clapfm-evc-high-fidelity-and-flexible","title":"ClapFM-EVC: High-Fidelity and Flexible Emotional Voice Conversion with Dual Control from Natural Language and Speech","date":"2025-05-20","arxiv_id":"2505.13805","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-self-supervised-features-for","title":"Investigating self-supervised features for expressive, multilingual voice conversion","date":"2025-05-13","arxiv_id":"2505.08278","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-optimal-transport-and-voice","title":"Discrete Optimal Transport and Voice Conversion","date":"2025-05-07","arxiv_id":"2505.04382","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-network-based-voice","title":"Generative Adversarial Network based Voice Conversion: Techniques, Challenges, and Recent Advancements","date":"2025-04-27","arxiv_id":"2504.19197","repositories_listed":0,"syntology":null},{"url":null,"slug":"fadel-uncertainty-aware-fake-audio-detection","title":"FADEL: Uncertainty-aware Fake Audio Detection with Evidential Deep Learning","date":"2025-04-22","arxiv_id":"2504.15663","repositories_listed":0,"syntology":null},{"url":null,"slug":"collective-learning-mechanism-based-optimal","title":"Collective Learning Mechanism based Optimal Transport Generative Adversarial Network for Non-parallel Voice Conversion","date":"2025-04-18","arxiv_id":"2504.13791","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-with-diverse-intonation","title":"Voice Conversion with Diverse Intonation using Conditional Variational Auto-Encoder","date":"2025-04-16","arxiv_id":"2504.12005","repositories_listed":0,"syntology":null}],"record_sha256":"317fbe89fea9a4dc00e3dd68aa275d48b94d448d865c68ad10b786e3d5e6c0d6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}