{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/voice-conversion/papers/4","list_of":"/task/voice-conversion","task":"Voice Conversion","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":6,"rows_per_page":100,"rows":[301,400],"of":520,"counts":{"archive_papers_tagged":520,"with_a_code_link":175,"where_syntology_ran_a_sample":41,"not_listed_spam_title":0,"listed":520,"listed_where_code_ran":41,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":32,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":32,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/voice-conversion","prev":"/task/voice-conversion/papers/3","next":"/task/voice-conversion/papers/5","papers":[{"url":null,"slug":"face-driven-zero-shot-voice-conversion-with","title":"Face-Driven Zero-Shot Voice Conversion with Memory-based Face-Voice Alignment","date":"2023-09-18","arxiv_id":"2309.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptvc-flexible-stylistic-voice-conversion","title":"PromptVC: Flexible Stylistic Voice Conversion in Latent Space Driven by Natural Language Prompts","date":"2023-09-17","arxiv_id":"2309.09262","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-knowledge-distillation-via-flow","title":"Cross-lingual Knowledge Distillation via Flow-based Voice Conversion for Robust Polyglot Text-To-Speech","date":"2023-09-15","arxiv_id":"2309.08255","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-voice-conversion-for-dissimilar","title":"Improving Voice Conversion for Dissimilar Speakers Using Perceptual Losses","date":"2023-09-15","arxiv_id":"2309.08263","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-and-limited-data-voice-conversion","title":"Parallel and Limited Data Voice Conversion Using Stochastic Variational Deep Kernel Learning","date":"2023-09-08","arxiv_id":"2309.04420","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylebook-content-dependent-speaking-style","title":"Stylebook: Content-Dependent Speaking Style Modeling for Any-to-Any Voice Conversion using Only Speech Data","date":"2023-09-06","arxiv_id":"2309.02730","repositories_listed":0,"syntology":null},{"url":null,"slug":"msm-vc-high-fidelity-source-style-transfer","title":"MSM-VC: High-fidelity Source Style Transfer for Non-Parallel Voice Conversion by Multi-scale Style Modeling","date":"2023-09-03","arxiv_id":"2309.01142","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpsp-learning-speech-concepts-from-phoneme","title":"Learning Speech Representation From Contrastive Token-Acoustic Pretraining","date":"2023-09-01","arxiv_id":"2309.00424","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-zero-shot-speaker-adaptive","title":"Generalizable Zero-Shot Speaker Adaptive Speech Synthesis with Disentangled Representations","date":"2023-08-24","arxiv_id":"2308.13007","repositories_listed":0,"syntology":null},{"url":"/paper/real-time-detection-of-ai-generated-speech","slug":"real-time-detection-of-ai-generated-speech","title":"Real-time Detection of AI-Generated Speech for DeepFake Voice Conversion","date":"2023-08-24","arxiv_id":"2308.12734","repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-convolutional-autoencoder","title":"Effects of Convolutional Autoencoder Bottleneck Width on StarGAN-based Singing Technique Conversion","date":"2023-08-19","arxiv_id":"2308.10021","repositories_listed":0,"syntology":null},{"url":null,"slug":"slmgan-exploiting-speech-language-model","title":"SLMGAN: Exploiting Speech Language Model Representations for Unsupervised Zero-Shot Voice Conversion in GANs","date":"2023-07-18","arxiv_id":"2307.09435","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-f0-synthesis-for-speaker","title":"Deep Learning-based F0 Synthesis for Speaker Anonymization","date":"2023-06-29","arxiv_id":"2306.16860","repositories_listed":0,"syntology":null},{"url":null,"slug":"fake-the-real-backdoor-attack-on-deep-speech","title":"Fake the Real: Backdoor Attack on Deep Speech Classification via Voice Conversion","date":"2023-06-28","arxiv_id":"2306.15875","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-voice-anonymization-for-enhanced","title":"Two-Stage Voice Anonymization for Enhanced Privacy","date":"2023-06-28","arxiv_id":"2306.16069","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-disentanglement-for-voice","title":"Automatic Speech Disentanglement for Voice Conversion using Rank Module and Speech Augmentation","date":"2023-06-21","arxiv_id":"2306.12259","repositories_listed":0,"syntology":null},{"url":null,"slug":"lm-vc-zero-shot-voice-conversion-via-speech","title":"LM-VC: Zero-shot Voice Conversion via Speech Generation based on Language Models","date":"2023-06-18","arxiv_id":"2306.10521","repositories_listed":0,"syntology":null},{"url":null,"slug":"alo-vc-any-to-any-low-latency-one-shot-voice","title":"ALO-VC: Any-to-any Low-latency One-shot Voice Conversion","date":"2023-06-01","arxiv_id":"2306.01100","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-a-voice-unified-voice-synthesis-with","title":"Make-A-Voice: Unified Voice Synthesis With Discrete Representation","date":"2023-05-30","arxiv_id":"2305.19269","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-personalized-synthetic-voices-from","title":"Creating Personalized Synthetic Voices from Post-Glossectomy Speech with Guided Diffusion Models","date":"2023-05-27","arxiv_id":"2305.17436","repositories_listed":0,"syntology":null},{"url":null,"slug":"iteratively-improving-speech-recognition-and","title":"Iteratively Improving Speech Recognition and Voice Conversion","date":"2023-05-24","arxiv_id":"2305.15055","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualvc-dual-mode-voice-conversion-using-intra","title":"DualVC: Dual-mode Voice Conversion using Intra-model Knowledge Distillation and Hybrid Predictive Coding","date":"2023-05-21","arxiv_id":"2305.12425","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-diverse-voice","title":"Data Augmentation for Diverse Voice Conversion in Noisy Environments","date":"2023-05-18","arxiv_id":"2305.10684","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-speaker-disentanglement-using","title":"Adversarial Speaker Disentanglement Using Unannotated External Data for Self-supervised Representation Based Voice Conversion","date":"2023-05-16","arxiv_id":"2305.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-temporal-channel-speaker","title":"Multi-level Temporal-channel Speaker Retrieval for Zero-shot Voice Conversion","date":"2023-05-12","arxiv_id":"2305.07204","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignsts-speech-to-singing-conversion-via","title":"AlignSTS: Speech-to-Singing Conversion via Cross-Modal Alignment","date":"2023-05-08","arxiv_id":"2305.04476","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-speaker-anonymization-on","title":"Evaluation of Speaker Anonymization on Emotional Speech","date":"2023-04-15","arxiv_id":"2305.01759","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-representations-for-singing","title":"Self-Supervised Representations for Singing Voice Conversion","date":"2023-03-21","arxiv_id":"2303.12197","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-latent-regressor","title":"A Comparative Analysis Of Latent Regressor Losses For Singing Voice Conversion","date":"2023-02-27","arxiv_id":"2302.13678","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-face-and-voice-style-transfer","title":"Cross-modal Face- and Voice-style Transfer","date":"2023-02-27","arxiv_id":"2302.13838","repositories_listed":0,"syntology":null},{"url":null,"slug":"catch-you-and-i-can-revealing-source","title":"Catch You and I Can: Revealing Source Voiceprint Against Voice Conversion","date":"2023-02-24","arxiv_id":"2302.12434","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonparallel-emotional-voice-conversion-for","title":"Nonparallel Emotional Voice Conversion For Unseen Speaker-Emotion Pairs Using Dual Domain Adversarial Network & Virtual Domain Pairing","date":"2023-02-21","arxiv_id":"2302.10536","repositories_listed":0,"syntology":null},{"url":null,"slug":"ace-vc-adaptive-and-controllable-voice","title":"ACE-VC: Adaptive and Controllable Voice Conversion using Explicitly Disentangled Self-supervised Speech Representations","date":"2023-02-16","arxiv_id":"2302.08137","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-low-resource-accents-without-accent","title":"Modelling low-resource accents without accent-specific TTS frontend","date":"2023-01-11","arxiv_id":"2301.04606","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifyspeech-a-unified-framework-for-zero-shot","title":"UnifySpeech: A Unified Framework for Zero-shot Text-to-Speech and Voice Conversion","date":"2023-01-10","arxiv_id":"2301.03801","repositories_listed":0,"syntology":null},{"url":null,"slug":"vsvc-backdoor-attack-against-keyword-spotting","title":"VSVC: Backdoor attack against Keyword Spotting based on Voiceprint Selection and Voice Conversion","date":"2022-12-20","arxiv_id":"2212.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-prosody-representations-with","title":"Disentangling Prosody Representations with Unsupervised Speech Reconstruction","date":"2022-12-14","arxiv_id":"2212.06972","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-feature-learning-for-real-time","title":"Disentangled Feature Learning for Real-Time Neural Speech Coding","date":"2022-11-22","arxiv_id":"2211.11960","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-anti-spoofing-using-a-simple-attention","title":"Audio Anti-spoofing Using a Simple Attention Module and Joint Optimization Based on Additive Angular Margin Loss and Meta-learning","date":"2022-11-17","arxiv_id":"2211.09898","repositories_listed":0,"syntology":null},{"url":null,"slug":"delivering-speaking-style-in-low-resource","title":"Delivering Speaking Style in Low-resource Voice Conversion with Multi-factor Constraints","date":"2022-11-16","arxiv_id":"2211.08857","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-disentangled-speech-representations","title":"Improved disentangled speech representations using contrastive learning in factorized hierarchical variational autoencoder","date":"2022-11-15","arxiv_id":"2211.08191","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-vc-highly-expressive-voice","title":"Expressive-VC: Highly Expressive Voice Conversion with Attention Fusion of Bottleneck and Perturbation Features","date":"2022-11-09","arxiv_id":"2211.04710","repositories_listed":0,"syntology":null},{"url":null,"slug":"preserving-background-sound-in-noise-robust","title":"Preserving background sound in noise-robust voice conversion via multi-task learning","date":"2022-11-06","arxiv_id":"2211.03036","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-automatic-speaker-verification-and","title":"Combining Automatic Speaker Verification and Prosody Analysis for Synthetic Speech Detection","date":"2022-10-31","arxiv_id":"2210.17222","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-text-to-speech-with-flow-based","title":"Cross-lingual Text-To-Speech with Flow-based Voice Conversion for Improved Pronunciation","date":"2022-10-31","arxiv_id":"2210.17264","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-voice-conversion-via-intermediate","title":"Streaming Voice Conversion Via Intermediate Bottleneck Features And Non-streaming Teacher Guidance","date":"2022-10-27","arxiv_id":"2210.15158","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-cloak-intelligibility-naturalness-timbre","title":"V-Cloak: Intelligibility-, Naturalness- & Timbre-Preserving Real-Time Voice Anonymization","date":"2022-10-27","arxiv_id":"2210.15140","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-speech-representation-learning-1","title":"Disentangled Speech Representation Learning for One-Shot Cross-lingual Voice Conversion Using $β$-VAE","date":"2022-10-25","arxiv_id":"2210.13771","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaspeech-speech-effects-switch-along-with","title":"MetaSpeech: Speech Effects Switch Along with Environment for Metaverse","date":"2022-10-25","arxiv_id":"2210.13811","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-emotion-modelling-for-emotional-voice","title":"Mixed-EVC: Mixed Emotion Synthesis and Control in Voice Conversion","date":"2022-10-25","arxiv_id":"2210.13756","repositories_listed":0,"syntology":null},{"url":null,"slug":"disc-vc-disentangled-and-f0-controllable","title":"DisC-VC: Disentangled and F0-Controllable Neural Voice Conversion","date":"2022-10-20","arxiv_id":"2210.11059","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-one-shot-singing-voice-conversion","title":"Robust One-Shot Singing Voice Conversion","date":"2022-10-20","arxiv_id":"2210.11096","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-star-gans-for-voice-conversion-with","title":"Boosting Star-GANs for Voice Conversion with Contrastive Discriminator","date":"2022-09-21","arxiv_id":"2209.10088","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-parallel-voice-conversion-for-asr","title":"Non-Parallel Voice Conversion for ASR Augmentation","date":"2022-09-15","arxiv_id":"2209.06987","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-rater-and-system-metadata-to-explain","title":"Using Rater and System Metadata to Explain Variance in the VoiceMOS Challenge 2022 Dataset","date":"2022-09-14","arxiv_id":"2209.06358","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-into-target-speaking-rate","title":"Investigation into Target Speaking Rate Adaptation for Voice Conversion","date":"2022-09-05","arxiv_id":"2209.01978","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-disentangled-representations-all-you-need","title":"Are disentangled representations all you need to build speaker anonymization systems?","date":"2022-08-22","arxiv_id":"2208.10497","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-world-synthesizer-based-neural","title":"Differentiable WORLD Synthesizer-based Neural Vocoder With Application To End-To-End Audio Style Transfer","date":"2022-08-15","arxiv_id":"2208.07282","repositories_listed":0,"syntology":null},{"url":null,"slug":"tgavc-improving-autoencoder-voice-conversion","title":"TGAVC: Improving Autoencoder Voice Conversion with Text-Guided and Adversarial Training","date":"2022-08-08","arxiv_id":"2208.04035","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-data-no-problem-low-resource-language","title":"Low-data? No problem: low-resource, language-agnostic conversational text-to-speech via F0-conditioned data augmentation","date":"2022-07-29","arxiv_id":"2207.14607","repositories_listed":0,"syntology":null},{"url":null,"slug":"transplantation-of-conversational-speaking","title":"Transplantation of Conversational Speaking Style with Interjections in Sequence-to-Sequence Speech Synthesis","date":"2022-07-25","arxiv_id":"2207.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"glowvc-mel-spectrogram-space-disentangling","title":"GlowVC: Mel-spectrogram space disentangling model for language-independent text-free voice conversion","date":"2022-07-04","arxiv_id":"2207.01454","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-speaker-representation","title":"A Hierarchical Speaker Representation Framework for One-shot Singing Voice Conversion","date":"2022-06-28","arxiv_id":"2206.13762","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-speech-representations-for-the","title":"Comparison of Speech Representations for the MOS Prediction System","date":"2022-06-28","arxiv_id":"2206.13817","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-source-speakers-for-voice","title":"Identifying Source Speakers for Voice Conversion based Spoofing Attacks on Speaker Verification Systems","date":"2022-06-18","arxiv_id":"2206.09103","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-voice-conversion-with-information","title":"End-to-End Voice Conversion with Information Perturbation","date":"2022-06-15","arxiv_id":"2206.07569","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-dubbing-lip-synchronous-voice-preserving","title":"Face-Dubbing++: Lip-Synchronous, Voice Preserving Translation of Videos","date":"2022-06-09","arxiv_id":"2206.04523","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-inter-and-intra-speaker-voice","title":"Investigating Inter- and Intra-speaker Voice Conversion using Audiobooks","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attentive-activation-function-for-improving","title":"Attentive activation function for improving end-to-end spoofing countermeasure systems","date":"2022-05-03","arxiv_id":"2205.01528","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-speaker-emotion-transfer-for-low","title":"Cross-Speaker Emotion Transfer for Low-Resource Text-to-Speech Using Non-Parallel Voice Conversion with Pitch-Shift Data Augmentation","date":"2022-04-21","arxiv_id":"2204.10020","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-deep-fake-detection-system-with-neural","title":"Audio Deep Fake Detection System with Neural Stitching for ADD 2022","date":"2022-04-19","arxiv_id":"2204.08720","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-domain-adversarial-voice-conversion-for","title":"Time Domain Adversarial Voice Conversion for ADD 2022","date":"2022-04-19","arxiv_id":"2204.08692","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-partialspoof-database-and-countermeasures","title":"The PartialSpoof Database and Countermeasures for the Detection of Short Fake Speech Segments Embedded in an Utterance","date":"2022-04-11","arxiv_id":"2204.05177","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-selective-self-distillation","title":"Representation Selective Self-distillation and wav2vec 2.0 Feature Exploration for Spoof-aware Speaker Verification","date":"2022-04-06","arxiv_id":"2204.02639","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-speech-representation-learning","title":"Disentangled Speech Representation Learning Based on Factorized Hierarchical Variational Autoencoder with Self-Supervised Objective","date":"2022-04-05","arxiv_id":"2204.02166","repositories_listed":0,"syntology":null},{"url":null,"slug":"anti-spoofing-using-transfer-learning-with","title":"Anti-Spoofing Using Transfer Learning with Variational Information Bottleneck","date":"2022-04-04","arxiv_id":"2204.01387","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-speech-representations","title":"Self-Supervised Speech Representations Preserve Speech Characteristics while Anonymizing Voices","date":"2022-04-04","arxiv_id":"2204.01677","repositories_listed":0,"syntology":null},{"url":null,"slug":"wavthruvec-latent-speech-representation-as","title":"WavThruVec: Latent speech representation as intermediate features for neural speech synthesis","date":"2022-03-31","arxiv_id":"2203.16930","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-zero-shot-many-to-many-voice","title":"Enhancing Zero-Shot Many to Many Voice Conversion with Self-Attention VAE","date":"2022-03-30","arxiv_id":"2203.16037","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-analysis-of-sequence-to-sequence","title":"An Overview & Analysis of Sequence-to-Sequence Emotional Voice Conversion","date":"2022-03-29","arxiv_id":"2203.15873","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-voice-conversion-and-code","title":"Analysis of Voice Conversion and Code-Switching Synthesis Using VQ-VAE","date":"2022-03-28","arxiv_id":"2203.14640","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangleing-content-and-fine-grained","title":"Disentangleing Content and Fine-grained Prosody Information via Hybrid ASR Bottleneck Features for Voice Conversion","date":"2022-03-24","arxiv_id":"2203.12813","repositories_listed":0,"syntology":null},{"url":null,"slug":"separating-content-from-speaker-identity-in","title":"Separating Content from Speaker Identity in Speech for the Assessment of Cognitive Impairments","date":"2022-03-21","arxiv_id":"2203.10827","repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-few-shot-voice-cloning-using-multi","title":"Improve few-shot voice cloning using multi-modal learning","date":"2022-03-18","arxiv_id":"2203.09708","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-free-non-parallel-many-to-many-voice","title":"Text-free non-parallel many-to-many voice conversion using normalising flows","date":"2022-03-15","arxiv_id":"2203.08009","repositories_listed":0,"syntology":null},{"url":null,"slug":"vcvts-multi-speaker-video-to-speech-synthesis","title":"VCVTS: Multi-speaker Video-to-Speech synthesis via cross-modal knowledge transfer from voice conversion","date":"2022-02-18","arxiv_id":"2202.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-filter-few-shot-text-to-speech-speaker","title":"Voice Filter: Few-shot text-to-speech speaker adaptation using voice conversion as a post-processing module","date":"2022-02-16","arxiv_id":"2202.08164","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-fake-audio-detection-by-self","title":"Partially Fake Audio Detection by Self-attention-based Fake Span Discovery","date":"2022-02-14","arxiv_id":"2202.06684","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-speaker-style-transfer-for-text-to","title":"Cross-speaker style transfer for text-to-speech using data augmentation","date":"2022-02-10","arxiv_id":"2202.05083","repositories_listed":0,"syntology":null},{"url":null,"slug":"invertible-voice-conversion","title":"Invertible Voice Conversion","date":"2022-01-26","arxiv_id":"2201.10687","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-effectiveness-of-time-stretching-for","title":"The Effectiveness of Time Stretching for Enhancing Dysarthric Speech for Improved Dysarthric Speech Recognition","date":"2022-01-13","arxiv_id":"2201.04908","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotion-intensity-and-its-control-for","title":"Emotion Intensity and its Control for Emotional Voice Conversion","date":"2022-01-10","arxiv_id":"2201.03967","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-transformation-of-spoofing","title":"Adversarial Transformation of Spoofing Attacks for Voice Biometrics","date":"2022-01-04","arxiv_id":"2201.01226","repositories_listed":0,"syntology":null},{"url":null,"slug":"iqdubbing-prosody-modeling-based-on-discrete","title":"IQDUBBING: Prosody modeling based on discrete self-supervised speech representation for expressive voice conversion","date":"2022-01-02","arxiv_id":"2201.00269","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-exploitation-of-multiple-feature","title":"The exploitation of Multiple Feature Extraction Techniques for Speaker Identification in Emotional States under Disguised Voices","date":"2021-12-15","arxiv_id":"2112.07940","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-robust-zero-shot-voice-conversion","title":"Training Robust Zero-Shot Voice Conversion Models with Self-supervised Features","date":"2021-12-08","arxiv_id":"2112.04424","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-deep-hierarchical-variational","title":"Conditional Deep Hierarchical Variational Autoencoder for Voice Conversion","date":"2021-12-06","arxiv_id":"2112.02796","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicemixer-adversarial-voice-style-mixup","title":"VoiceMixer: Adversarial Voice Style Mixup","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-voice-conversion-for-style-transfer","title":"One-shot Voice Conversion For Style Transfer Based On Speaker Adaptation","date":"2021-11-24","arxiv_id":"2111.12277","repositories_listed":0,"syntology":null},{"url":null,"slug":"ac-vc-non-parallel-low-latency-phonetic","title":"AC-VC: Non-parallel Low Latency Phonetic Posteriorgrams Based Voice Conversion","date":"2021-11-12","arxiv_id":"2111.06601","repositories_listed":0,"syntology":null}],"record_sha256":"34483d6fcc0ca6a6125278646d7098a0fd84fe608ac2f92e7ed25713b7b44ec3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}