{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/voice-conversion/papers/3","list_of":"/task/voice-conversion","task":"Voice Conversion","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":6,"rows_per_page":100,"rows":[201,300],"of":520,"counts":{"archive_papers_tagged":520,"with_a_code_link":175,"where_syntology_ran_a_sample":41,"not_listed_spam_title":0,"listed":520,"listed_where_code_ran":41,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":32,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":32,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/voice-conversion","prev":"/task/voice-conversion/papers/2","next":"/task/voice-conversion/papers/4","papers":[{"url":null,"slug":"mitigating-timbre-leakage-with-universal","title":"Mitigating Timbre Leakage with Universal Semantic Mapping Residual Block for Voice Conversion","date":"2025-04-11","arxiv_id":"2504.08524","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exhaustive-evaluation-of-tts-and-vc-based","title":"An Exhaustive Evaluation of TTS- and VC-based Data Augmentation for ASR","date":"2025-03-11","arxiv_id":"2503.08954","repositories_listed":0,"syntology":null},{"url":null,"slug":"asvspoof-5-design-collection-and-validation","title":"ASVspoof 5: Design, Collection and Validation of Resources for Spoofing, Deepfake, and Adversarial Attack Detection Using Crowdsourced Speech","date":"2025-02-13","arxiv_id":"2502.08857","repositories_listed":0,"syntology":null},{"url":null,"slug":"vevo-controllable-zero-shot-voice-imitation","title":"Vevo: Controllable Zero-Shot Voice Imitation with Self-Supervised Disentanglement","date":"2025-02-11","arxiv_id":"2502.07243","repositories_listed":0,"syntology":null},{"url":null,"slug":"singing-voice-conversion-with-accompaniment","title":"Singing Voice Conversion with Accompaniment Using Self-Supervised Representation-Based Melody Features","date":"2025-02-07","arxiv_id":"2502.04722","repositories_listed":0,"syntology":null},{"url":null,"slug":"focalcodec-low-bitrate-speech-coding-via","title":"FocalCodec: Low-Bitrate Speech Coding via Focal Modulation Networks","date":"2025-02-06","arxiv_id":"2502.04465","repositories_listed":0,"syntology":null},{"url":null,"slug":"genvc-self-supervised-zero-shot-voice","title":"GenVC: Self-Supervised Zero-Shot Voice Conversion","date":"2025-02-06","arxiv_id":"2502.04519","repositories_listed":0,"syntology":null},{"url":null,"slug":"voiceprompter-robust-zero-shot-voice","title":"VoicePrompter: Robust Zero-Shot Voice Conversion with Voice Prompt and Conditional Flow Matching","date":"2025-01-29","arxiv_id":"2501.17612","repositories_listed":0,"syntology":null},{"url":null,"slug":"stepback-enhanced-disentanglement-for-voice","title":"Stepback: Enhanced Disentanglement for Voice Conversion via Multi-Task Learning","date":"2025-01-26","arxiv_id":"2501.15613","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-audio-deepfake-detection-via","title":"Generalizable Audio Deepfake Detection via Latent Space Refinement and Augmentation","date":"2025-01-24","arxiv_id":"2501.14240","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-rhythm-and-voice-conversion-of","title":"Unsupervised Rhythm and Voice Conversion of Dysarthric to Healthy Speech for ASR","date":"2025-01-17","arxiv_id":"2501.10256","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-along-perceptual-voice","title":"Speech Synthesis along Perceptual Voice Quality Dimensions","date":"2025-01-15","arxiv_id":"2501.08791","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-for-automatically","title":"Speech Recognition for Automatically Assessing Afrikaans and isiXhosa Preschool Oral Narratives","date":"2025-01-11","arxiv_id":"2501.06478","repositories_listed":0,"syntology":null},{"url":null,"slug":"zsvc-zero-shot-style-voice-conversion-with","title":"ZSVC: Zero-shot Style Voice Conversion with Disentangled Latent Diffusion Models and Adversarial Training","date":"2025-01-08","arxiv_id":"2501.04416","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-and-detecting-various-types-of","title":"Generating and Detecting Various Types of Fake Image and Audio Content: A Review of Modern Deep Learning Technologies and Tools","date":"2025-01-07","arxiv_id":"2501.06227","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptvc-high-quality-voice-conversion-with","title":"AdaptVC: High Quality Voice Conversion with Adaptive Learning","date":"2025-01-02","arxiv_id":"2501.01347","repositories_listed":0,"syntology":null},{"url":null,"slug":"emoreg-directional-latent-vector-modeling-for","title":"EmoReg: Directional Latent Vector Modeling for Emotional Intensity Regularization in Diffusion-based Voice Conversion","date":"2024-12-29","arxiv_id":"2412.20359","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-model-for-voice-and-accent","title":"A Unified Model For Voice and Accent Conversion In Speech and Singing using Self-Supervised Learning and Feature Extraction","date":"2024-12-11","arxiv_id":"2412.08312","repositories_listed":0,"syntology":null},{"url":null,"slug":"stablevc-style-controllable-zero-shot-voice","title":"StableVC: Style Controllable Zero-Shot Voice Conversion with Conditional Flow Matching","date":"2024-12-06","arxiv_id":"2412.04724","repositories_listed":0,"syntology":null},{"url":null,"slug":"noro-a-noise-robust-one-shot-voice-conversion","title":"Noro: A Noise-Robust One-shot Voice Conversion System with Hidden Speaker Representation Capabilities","date":"2024-11-29","arxiv_id":"2411.19770","repositories_listed":0,"syntology":null},{"url":null,"slug":"skqvc-one-shot-voice-conversion-by-k-means","title":"SKQVC: One-Shot Voice Conversion by K-Means Quantization with Self-Supervised Speech Representations","date":"2024-11-25","arxiv_id":"2411.16147","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctefm-vc-zero-shot-voice-conversion-based-on","title":"CTEFM-VC: Zero-Shot Voice Conversion Based on Content-Aware Timbre Ensemble Modeling and Flow Matching","date":"2024-11-04","arxiv_id":"2411.02026","repositories_listed":0,"syntology":null},{"url":null,"slug":"lscodec-low-bitrate-and-speaker-decoupled","title":"LSCodec: Low-Bitrate and Speaker-Decoupled Discrete Speech Codec","date":"2024-10-21","arxiv_id":"2410.15764","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-voice-quality-in-speech","title":"Improving Voice Quality in Speech Anonymization With Just Perception-Informed Losses","date":"2024-10-20","arxiv_id":"2410.15499","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pilot-study-of-applying-sequence-to","title":"A Pilot Study of Applying Sequence-to-Sequence Voice Conversion to Evaluate the Intelligibility of L2 Speech Using a Native Speaker's Shadowings","date":"2024-10-03","arxiv_id":"2410.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"takin-vc-zero-shot-voice-conversion-via","title":"Takin-VC: Expressive Zero-Shot Voice Conversion via Adaptive Hybrid Content Encoding and Enhanced Timbre Modeling","date":"2024-10-02","arxiv_id":"2410.01350","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-synthetic-data-for-cross-speaker","title":"Exploring synthetic data for cross-speaker style transfer in style representation based TTS","date":"2024-09-25","arxiv_id":"2409.17364","repositories_listed":0,"syntology":null},{"url":null,"slug":"textless-nlp-zero-resource-challenge-with-low","title":"Textless NLP -- Zero Resource Challenge with Low Resource Compute","date":"2024-09-24","arxiv_id":"2409.19015","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-unit-based-masking-for-improving","title":"Discrete Unit based Masking for Improving Disentanglement in Voice Conversion","date":"2024-09-17","arxiv_id":"2409.11560","repositories_listed":0,"syntology":null},{"url":null,"slug":"hltcoe-jhu-submission-to-the-voice-privacy","title":"HLTCOE JHU Submission to the Voice Privacy Challenge 2024","date":"2024-09-13","arxiv_id":"2409.08913","repositories_listed":0,"syntology":null},{"url":null,"slug":"lhq-svc-lightweight-and-high-quality-singing","title":"LHQ-SVC: Lightweight and High Quality Singing Voice Conversion Modeling","date":"2024-09-13","arxiv_id":"2409.08583","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-captcha-a-study-of-resilience-of-deepfake","title":"D-CAPTCHA++: A Study of Resilience of Deepfake CAPTCHA under Transferable Imperceptible Adversarial Attack","date":"2024-09-11","arxiv_id":"2409.07390","repositories_listed":0,"syntology":null},{"url":null,"slug":"vc-enhance-speech-restoration-with-integrated","title":"VC-ENHANCE: Speech Restoration with Integrated Noise Suppression and Voice Conversion","date":"2024-09-10","arxiv_id":"2409.06126","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicewukong-benchmarking-deepfake-voice","title":"VoiceWukong: Benchmarking Deepfake Voice Detection","date":"2024-09-10","arxiv_id":"2409.06348","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffevc-any-to-any-emotion-voice-conversion","title":"ZSDEVC: Zero-Shot Diffusion-based Emotional Voice Conversion with Disentangled Mechanism","date":"2024-09-05","arxiv_id":"2409.03636","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-and-style-disentanglement-of-speech","title":"Speaker and Style Disentanglement of Speech Based on Contrastive Predictive Coding Supported Factorized Variational Autoencoder","date":"2024-09-05","arxiv_id":"2409.03520","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastvoicegrad-one-step-diffusion-based-voice","title":"FastVoiceGrad: One-step Diffusion-Based Voice Conversion with Adversarial Conditional Diffusion Distillation","date":"2024-09-03","arxiv_id":"2409.02245","repositories_listed":0,"syntology":null},{"url":null,"slug":"pureformer-vc-non-parallel-one-shot-voice","title":"Pureformer-VC: Non-parallel One-Shot Voice Conversion with Pure Transformer Blocks and Triplet Discriminative Training","date":"2024-09-03","arxiv_id":"2409.01668","repositories_listed":0,"syntology":null},{"url":null,"slug":"ustc-kxdigit-system-description-for-asvspoof5","title":"USTC-KXDIGIT System Description for ASVspoof5 Challenge","date":"2024-09-03","arxiv_id":"2409.01695","repositories_listed":0,"syntology":null},{"url":null,"slug":"vec2wav-2-0-advancing-voice-conversion-via","title":"vec2wav 2.0: Advancing Voice Conversion via Discrete Token Vocoders","date":"2024-09-03","arxiv_id":"2409.01995","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-your-speech-style-a-novel-zero-shot","title":"Seeing Your Speech Style: A Novel Zero-Shot Identity-Disentanglement Face-based Voice Conversion","date":"2024-09-01","arxiv_id":"2409.00700","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-residual-extraction-based-pre","title":"Progressive Residual Extraction based Pre-training for Speech Representation Learning","date":"2024-08-31","arxiv_id":"2409.00387","repositories_listed":0,"syntology":null},{"url":null,"slug":"aasist3-kan-enhanced-aasist-speech-deepfake","title":"AASIST3: KAN-Enhanced AASIST Speech Deepfake Detection using SSL Features and Additional Regularization for the ASVspoof 2024 Challenge","date":"2024-08-30","arxiv_id":"2408.17352","repositories_listed":0,"syntology":null},{"url":null,"slug":"emoattack-utilizing-emotional-voice","title":"EmoAttack: Utilizing Emotional Voice Conversion for Speech Backdoor Attacks on Deep Speech Classification Models","date":"2024-08-28","arxiv_id":"2408.15508","repositories_listed":0,"syntology":null},{"url":null,"slug":"maskcyclegan-based-whisper-to-normal-speech","title":"MaskCycleGAN-based Whisper to Normal Speech Conversion","date":"2024-08-27","arxiv_id":"2408.14797","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-improving-synthetic-audio-spoofing","title":"Toward Improving Synthetic Audio Spoofing Detection Robustness via Meta-Learning and Disentangled Training With Adversarial Examples","date":"2024-08-23","arxiv_id":"2408.13341","repositories_listed":0,"syntology":null},{"url":null,"slug":"lcm-svc-latent-diffusion-model-based-singing","title":"LCM-SVC: Latent Diffusion Model Based Singing Voice Conversion with Inference Acceleration via Latent Consistency Distillation","date":"2024-08-22","arxiv_id":"2408.12354","repositories_listed":0,"syntology":null},{"url":null,"slug":"vq-ctap-cross-modal-fine-grained-sequence","title":"VQ-CTAP: Cross-Modal Fine-Grained Sequence Representation Learning for Speech Processing","date":"2024-08-11","arxiv_id":"2408.05758","repositories_listed":0,"syntology":null},{"url":null,"slug":"mullivc-multi-lingual-voice-conversion-with","title":"MulliVC: Multi-lingual Voice Conversion With Cycle Consistency","date":"2024-08-08","arxiv_id":"2408.04708","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02712","title":"Automatic Voice Identification after Speech Resynthesis using PPG","date":"2024-08-05","arxiv_id":"2408.02712","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamvoice-evolving-into-end-to-end","title":"StreamVoice+: Evolving into End-to-end Streaming Zero-shot Voice Conversion","date":"2024-08-05","arxiv_id":"2408.02178","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-realistic-emotional-voice-conversion","title":"Towards Realistic Emotional Voice Conversion using Controllable Emotional Intensity","date":"2024-07-20","arxiv_id":"2407.14800","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-voiceprivacy-2022-challenge-progress-and","title":"The VoicePrivacy 2022 Challenge: Progress and Perspectives in Voice Anonymisation","date":"2024-07-16","arxiv_id":"2407.11516","repositories_listed":0,"syntology":null},{"url":null,"slug":"source-tracing-of-audio-deepfake-systems","title":"Source Tracing of Audio Deepfake Systems","date":"2024-07-10","arxiv_id":"2407.08016","repositories_listed":0,"syntology":null},{"url":null,"slug":"we-need-variations-in-speech-synthesis-sub","title":"We Need Variations in Speech Generation: Sub-center Modelling for Speaker Embeddings","date":"2024-07-05","arxiv_id":"2407.04291","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-asv-for-voice-identification","title":"Application of ASV for Voice Identification after VC and Duration Predictor Improvement in TTS Models","date":"2024-06-27","arxiv_id":"2406.19243","repositories_listed":0,"syntology":null},{"url":"/paper/dreamvoice-text-guided-voice-conversion","slug":"dreamvoice-text-guided-voice-conversion","title":"DreamVoice: Text-Guided Voice Conversion","date":"2024-06-24","arxiv_id":"2406.16314","repositories_listed":0,"syntology":null},{"url":null,"slug":"refxvc-cross-lingual-voice-conversion-with","title":"RefXVC: Cross-Lingual Voice Conversion with Enhanced Reference Leveraging","date":"2024-06-24","arxiv_id":"2406.16326","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualvc-3-leveraging-language-model-generated","title":"DualVC 3: Leveraging Language Model Generated Pseudo Context for End-to-end Low Latency Streaming Voice Conversion","date":"2024-06-12","arxiv_id":"2406.07846","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-child-speech-recognition-with","title":"Improving child speech recognition with augmented child-like speech","date":"2024-06-12","arxiv_id":"2406.10284","repositories_listed":0,"syntology":null},{"url":null,"slug":"svsnet-enhancing-speaker-voice-similarity","title":"SVSNet+: Enhancing Speaker Voice Similarity Assessment Models with Representations from Speech Foundation Models","date":"2024-06-12","arxiv_id":"2406.08445","repositories_listed":0,"syntology":null},{"url":null,"slug":"spa-svc-self-supervised-pitch-augmentation","title":"SPA-SVC: Self-supervised Pitch Augmentation for Singing Voice Conversion","date":"2024-06-09","arxiv_id":"2406.05692","repositories_listed":0,"syntology":null},{"url":null,"slug":"ldm-svc-latent-diffusion-model-based-zero","title":"LDM-SVC: Latent Diffusion Model Based Zero-Shot Any-to-Any Singing Voice Conversion with Singer Guidance","date":"2024-06-08","arxiv_id":"2406.05325","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-database-and-benchmark-for-source-speaker","title":"The Database and Benchmark for the Source Speaker Tracing Challenge 2024","date":"2024-06-07","arxiv_id":"2406.04951","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-naturalistic-voice-conversion","title":"Towards Naturalistic Voice Conversion: NaturalVoices Dataset with an Automatic Processing Pipeline","date":"2024-06-06","arxiv_id":"2406.04494","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-singing-voice-pre-training","title":"Self-Supervised Singing Voice Pre-Training towards Speech-to-Singing Conversion","date":"2024-06-04","arxiv_id":"2406.02429","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-and-accurate-zero-shot-high","title":"Real-Time and Accurate: Zero-shot High-Fidelity Singing Voice Conversion with Multi-Condition Flow Synthesis","date":"2024-05-23","arxiv_id":"2405.15093","repositories_listed":0,"syntology":null},{"url":null,"slug":"converting-anyone-s-voice-end-to-end","title":"Converting Anyone's Voice: End-to-End Expressive Voice Conversion with a Conditional Diffusion Model","date":"2024-05-02","arxiv_id":"2405.01730","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-is-authentic-speaker","title":"Who is Authentic Speaker","date":"2024-04-30","arxiv_id":"2405.00248","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-audio-deepfake-detection","title":"Retrieval-Augmented Audio Deepfake Detection","date":"2024-04-22","arxiv_id":"2404.13892","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptcodec-high-fidelity-neural-speech-codec","title":"PSCodec: A Series of High-Fidelity Low-bitrate Neural Speech Codecs Leveraging Prompt Encoders","date":"2024-04-03","arxiv_id":"2404.02702","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-augmentation-for-speaker","title":"Voice Conversion Augmentation for Speaker Recognition on Defective Datasets","date":"2024-04-01","arxiv_id":"2404.00863","repositories_listed":0,"syntology":null},{"url":null,"slug":"pavits-exploring-prosody-aware-vits-for-end","title":"PAVITS: Exploring Prosody-aware VITS for End-to-End Emotional Voice Conversion","date":"2024-03-03","arxiv_id":"2403.01494","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcription-and-translation-of-videos-using","title":"Transcription and translation of videos using fine-tuned XLSR Wav2Vec2 on custom dataset and mBART","date":"2024-03-01","arxiv_id":"2403.00212","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-the-stability-of-llm-based-speech","title":"Enhancing the Stability of LLM-based Speech Generation Systems through Self-Supervised Representations","date":"2024-02-05","arxiv_id":"2402.03407","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechcomposer-unifying-multiple-speech-tasks","title":"SpeechComposer: Unifying Multiple Speech Tasks with Prompt Composition","date":"2024-01-31","arxiv_id":"2401.18045","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-proactive-and-dual-prevention-mechanism","title":"SongBsAb: A Dual Prevention Approach against Singing Voice Conversion based Illegal Song Covers","date":"2024-01-30","arxiv_id":"2401.17133","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-speech-for-voice-privacy","title":"Adversarial speech for voice privacy protection from Personalized Speech generation","date":"2024-01-22","arxiv_id":"2401.11857","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamvoice-streamable-context-aware-language","title":"StreamVoice: Streamable Context-Aware Language Modeling for Real-time Zero-Shot Voice Conversion","date":"2024-01-19","arxiv_id":"2401.11053","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-the-linguistic-representations-from","title":"Transfer the linguistic representations from TTS to accent conversion with non-parallel data","date":"2024-01-07","arxiv_id":"2401.03538","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamvc-real-time-low-latency-voice","title":"StreamVC: Real-Time Low-Latency Voice Conversion","date":"2024-01-05","arxiv_id":"2401.03078","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-interactive-disentangling","title":"Attention-based Interactive Disentangling Network for Instance-level Emotional Voice Conversion","date":"2023-12-29","arxiv_id":"2312.17508","repositories_listed":0,"syntology":null},{"url":null,"slug":"ae-flow-autoencoder-normalizing-flow","title":"AE-Flow: AutoEncoder Normalizing Flow","date":"2023-12-27","arxiv_id":"2312.16552","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-data-augmentation-in-bias","title":"Exploring data augmentation in bias mitigation against non-native-accented speech","date":"2023-12-24","arxiv_id":"2312.15499","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-new-voices-using-normalizing-flows","title":"Creating New Voices using Normalizing Flows","date":"2023-12-22","arxiv_id":"2312.14569","repositories_listed":0,"syntology":null},{"url":null,"slug":"sef-vc-speaker-embedding-free-zero-shot-voice","title":"SEF-VC: Speaker Embedding Free Zero-Shot Voice Conversion with Cross Attention","date":"2023-12-14","arxiv_id":"2312.08676","repositories_listed":0,"syntology":null},{"url":null,"slug":"permod-perceptually-grounded-voice","title":"PerMod: Perceptually Grounded Voice Modification with Latent Diffusion Models","date":"2023-12-13","arxiv_id":"2312.08494","repositories_listed":0,"syntology":null},{"url":null,"slug":"vulnerability-of-automatic-identity","title":"Vulnerability of Automatic Identity Recognition to Audio-Visual Deepfakes","date":"2023-11-29","arxiv_id":"2311.17655","repositories_listed":0,"syntology":null},{"url":null,"slug":"custom-data-augmentation-for-low-resource-asr","title":"Custom Data Augmentation for low resource ASR using Bark and Retrieval-Based Voice Conversion","date":"2023-11-24","arxiv_id":"2311.14836","repositories_listed":0,"syntology":null},{"url":null,"slug":"reimagining-speech-a-scoping-review-of-deep","title":"Reimagining Speech: A Scoping Review of Deep Learning-Powered Voice Conversion","date":"2023-11-14","arxiv_id":"2311.08104","repositories_listed":0,"syntology":null},{"url":null,"slug":"parrot-trained-adversarial-examples-pushing","title":"Parrot-Trained Adversarial Examples: Pushing the Practicality of Black-Box Audio Attacks against Speaker Recognition Models","date":"2023-11-13","arxiv_id":"2311.07780","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-text-to-speech-systems-and","title":"An overview of text-to-speech systems and media applications","date":"2023-10-22","arxiv_id":"2310.14301","repositories_listed":0,"syntology":null},{"url":null,"slug":"selfvc-voice-conversion-with-iterative","title":"SelfVC: Voice Conversion With Iterative Refinement using Self Transformations","date":"2023-10-14","arxiv_id":"2310.09653","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-for-stuttered-speech","title":"Voice Conversion for Stuttered Speech, Instruments, Unseen Languages and Textually Described Voices","date":"2023-10-12","arxiv_id":"2310.08104","repositories_listed":0,"syntology":null},{"url":null,"slug":"autocycle-vc-towards-bottleneck-independent","title":"AutoCycle-VC: Towards Bottleneck-Independent Zero-Shot Cross-Lingual Voice Conversion","date":"2023-10-10","arxiv_id":"2310.06546","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-voice-conversion","title":"A Comparative Study of Voice Conversion Models with Large-Scale Speech and Singing Data: The T13 Systems for the Singing Voice Conversion Challenge 2023","date":"2023-10-08","arxiv_id":"2310.05203","repositories_listed":0,"syntology":null},{"url":null,"slug":"vits-based-singing-voice-conversion","title":"VITS-Based Singing Voice Conversion Leveraging Whisper and multi-scale F0 Modeling","date":"2023-10-04","arxiv_id":"2310.02802","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualvc-2-dynamic-masked-convolution-for","title":"DualVC 2: Dynamic Masked Convolution for Unified Streaming and Non-Streaming Voice Conversion","date":"2023-09-27","arxiv_id":"2309.15496","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-general-purpose-text-instruction","title":"Towards General-Purpose Text-Instruction-Guided Voice Conversion","date":"2023-09-25","arxiv_id":"2309.14324","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-silence-on-speech-anti-spoofing","title":"The Impact of Silence on Speech Anti-Spoofing","date":"2023-09-21","arxiv_id":"2309.11827","repositories_listed":0,"syntology":null}],"record_sha256":"95118c58eeebf7ef4e189c9adeeb808b2fbd6b281771cee4a542d498b2ed0d53","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}