{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/3","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":13,"rows_per_page":100,"rows":[201,300],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/2","next":"/task/speech-synthesis/papers/4","papers":[{"url":"/paper/take-the-hint-improving-arabic-diacritization","slug":"take-the-hint-improving-arabic-diacritization","title":"Take the Hint: Improving Arabic Diacritization with Partially-Diacritized Text","date":"2023-06-06","arxiv_id":"2306.03557","repositories_listed":1,"syntology":null},{"url":"/paper/why-we-should-report-the-details-in","slug":"why-we-should-report-the-details-in","title":"Why We Should Report the Details in Subjective Evaluation of TTS More Rigorously","date":"2023-06-03","arxiv_id":"2306.02044","repositories_listed":1,"syntology":null},{"url":"/paper/adaptermix-exploring-the-efficacy-of-mixture","slug":"adaptermix-exploring-the-efficacy-of-mixture","title":"ADAPTERMIX: Exploring the Efficacy of Mixture of Adapters for Low-Resource TTS Adaptation","date":"2023-05-29","arxiv_id":"2305.18028","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-tuning-of-loss-trade-offs-without","slug":"automatic-tuning-of-loss-trade-offs-without","title":"Automatic Tuning of Loss Trade-offs without Hyper-parameter Search in End-to-End Zero-Shot Speech Synthesis","date":"2023-05-26","arxiv_id":"2305.16699","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-text-to-speech-synthesis-for","slug":"multilingual-text-to-speech-synthesis-for","title":"Multilingual Text-to-Speech Synthesis for Turkic Languages Using Transliteration","date":"2023-05-25","arxiv_id":"2305.15749","repositories_listed":1,"syntology":null},{"url":"/paper/emns-imz-corpus-an-emotive-single-speaker","slug":"emns-imz-corpus-an-emotive-single-speaker","title":"EMNS /Imz/ Corpus: An emotive single-speaker dataset for narrative storytelling in games, television and graphic novels","date":"2023-05-22","arxiv_id":"2305.13137","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emns-imz-corpus-an-emotive-single-speaker#ran","syntology_url":"https://syntology.ai/paper/2305.13137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13137"}},"official":{"repos":["knoriy/emns-dct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-based-mel-spectrogram-enhancement","slug":"diffusion-based-mel-spectrogram-enhancement","title":"Diffusion-Based Mel-Spectrogram Enhancement for Personalized Speech Synthesis with Found Data","date":"2023-05-18","arxiv_id":"2305.10891","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-based-mel-spectrogram-enhancement#ran","syntology_url":"https://syntology.ai/paper/2305.10891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10891"}},"official":{"repos":["dmse4tts/dmse4tts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-generalization-ability-of","slug":"improving-generalization-ability-of","title":"Improving Generalization Ability of Countermeasures for New Mismatch Scenario by Combining Multiple Advanced Regularization Terms","date":"2023-05-18","arxiv_id":"2305.10940","repositories_listed":1,"syntology":null},{"url":"/paper/better-speech-synthesis-through-scaling","slug":"better-speech-synthesis-through-scaling","title":"Better speech synthesis through scaling","date":"2023-05-12","arxiv_id":"2305.07243","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-speech-synthesis-through-scaling#ran","syntology_url":"https://syntology.ai/paper/2305.07243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07243"}},"official":{"repos":["neonbjb/tortoise-tts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comospeech-one-step-speech-and-singing-voice","slug":"comospeech-one-step-speech-and-singing-voice","title":"CoMoSpeech: One-Step Speech and Singing Voice Synthesis via Consistency Model","date":"2023-05-11","arxiv_id":"2305.06908","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/comospeech-one-step-speech-and-singing-voice#ran","syntology_url":"https://syntology.ai/paper/2305.06908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06908"}},"official":{"repos":["zhenye234/CoMoSpeech"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bts-e-audio-deepfake-detection-using","slug":"bts-e-audio-deepfake-detection-using","title":"Bts-e: Audio deepfake detection using breathing-talking-silence encoder","date":"2023-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/source-filter-based-generative-adversarial","slug":"source-filter-based-generative-adversarial","title":"Source-Filter-Based Generative Adversarial Neural Vocoder for High Fidelity Speech Synthesis","date":"2023-04-26","arxiv_id":"2304.13270","repositories_listed":1,"syntology":null},{"url":"/paper/speak-foreign-languages-with-your-own-voice","slug":"speak-foreign-languages-with-your-own-voice","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","date":"2023-03-07","arxiv_id":"2303.03926","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speak-foreign-languages-with-your-own-voice#ran","syntology_url":"https://syntology.ai/paper/2303.03926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03926"}},"official":null}},{"url":"/paper/evaluating-parameter-efficient-transfer","slug":"evaluating-parameter-efficient-transfer","title":"Evaluating Parameter-Efficient Transfer Learning Approaches on SURE Benchmark for Speech Understanding","date":"2023-03-02","arxiv_id":"2303.03267","repositories_listed":1,"syntology":null},{"url":"/paper/imaginary-voice-face-styled-diffusion-model","slug":"imaginary-voice-face-styled-diffusion-model","title":"Imaginary Voice: Face-styled Diffusion Model for Text-to-Speech","date":"2023-02-27","arxiv_id":"2302.13700","repositories_listed":1,"syntology":null},{"url":"/paper/a-vector-quantized-approach-for-text-to","slug":"a-vector-quantized-approach-for-text-to","title":"A Vector Quantized Approach for Text to Speech Synthesis on Real-World Spontaneous Speech","date":"2023-02-08","arxiv_id":"2302.04215","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-vector-quantized-approach-for-text-to#ran","syntology_url":"https://syntology.ai/paper/2302.04215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04215"}},"official":{"repos":["b04901014/mqtts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-information-fusion-for-voice","slug":"cross-modal-information-fusion-for-voice","title":"Cross-modal information fusion for voice spoofing detection","date":"2023-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/time-out-of-mind-generating-emotionally","slug":"time-out-of-mind-generating-emotionally","title":"Time out of Mind: Generating Rate of Speech conditioned on emotion and speaker","date":"2023-01-29","arxiv_id":"2301.12331","repositories_listed":1,"syntology":null},{"url":"/paper/towards-voice-reconstruction-from-eeg-during","slug":"towards-voice-reconstruction-from-eeg-during","title":"Towards Voice Reconstruction from EEG during Imagined Speech","date":"2023-01-02","arxiv_id":"2301.07173","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-voice-reconstruction-from-eeg-during#ran","syntology_url":"https://syntology.ai/paper/2301.07173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07173"}},"official":{"repos":["youngeun1209/neurotalk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rwen-tts-relation-aware-word-encoding-network","slug":"rwen-tts-relation-aware-word-encoding-network","title":"RWEN-TTS: Relation-aware Word Encoding Network for Natural Text-to-Speech Synthesis","date":"2022-12-15","arxiv_id":"2212.07939","repositories_listed":1,"syntology":null},{"url":"/paper/mntts2-an-open-source-multi-speaker-mongolian","slug":"mntts2-an-open-source-multi-speaker-mongolian","title":"MnTTS2: An Open-Source Multi-Speaker Mongolian Text-to-Speech Synthesis Dataset","date":"2022-12-11","arxiv_id":"2301.00657","repositories_listed":1,"syntology":null},{"url":"/paper/videodubber-machine-translation-with-speech","slug":"videodubber-machine-translation-with-speech","title":"VideoDubber: Machine Translation with Speech-Aware Length Control for Video Dubbing","date":"2022-11-30","arxiv_id":"2211.16934","repositories_listed":1,"syntology":null},{"url":"/paper/prompttts-controllable-text-to-speech-with","slug":"prompttts-controllable-text-to-speech-with","title":"PromptTTS: Controllable Text-to-Speech with Text Descriptions","date":"2022-11-22","arxiv_id":"2211.12171","repositories_listed":1,"syntology":null},{"url":"/paper/embedding-a-differentiable-mel-cepstral","slug":"embedding-a-differentiable-mel-cepstral","title":"Embedding a Differentiable Mel-cepstral Synthesis Filter to a Neural Speech Synthesis System","date":"2022-11-21","arxiv_id":"2211.11222","repositories_listed":1,"syntology":null},{"url":"/paper/accented-text-to-speech-synthesis-with-a","slug":"accented-text-to-speech-synthesis-with-a","title":"Accented Text-to-Speech Synthesis with a Conditional Variational Autoencoder","date":"2022-11-07","arxiv_id":"2211.03316","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-for-speech-2","slug":"self-supervised-learning-for-speech-2","title":"Self-Supervised Learning for Speech Enhancement through Synthesis","date":"2022-11-04","arxiv_id":"2211.02542","repositories_listed":1,"syntology":null},{"url":"/paper/adapter-based-extension-of-multi-speaker-text","slug":"adapter-based-extension-of-multi-speaker-text","title":"Adapter-Based Extension of Multi-Speaker Text-to-Speech Model for New Speakers","date":"2022-11-01","arxiv_id":"2211.00585","repositories_listed":1,"syntology":null},{"url":"/paper/a-fast-and-accurate-pitch-estimation","slug":"a-fast-and-accurate-pitch-estimation","title":"A Fast and Accurate Pitch Estimation Algorithm Based on the Pseudo Wigner-Ville Distribution","date":"2022-10-27","arxiv_id":"2210.15272","repositories_listed":1,"syntology":null},{"url":"/paper/articulation-gan-unsupervised-modeling-of","slug":"articulation-gan-unsupervised-modeling-of","title":"Articulation GAN: Unsupervised modeling of articulatory learning","date":"2022-10-27","arxiv_id":"2210.15173","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-context-invariance-in-unsupervised","slug":"evaluating-context-invariance-in-unsupervised","title":"Evaluating context-invariance in unsupervised speech representations","date":"2022-10-27","arxiv_id":"2210.15775","repositories_listed":1,"syntology":null},{"url":"/paper/fctalker-fine-and-coarse-grained-context","slug":"fctalker-fine-and-coarse-grained-context","title":"FCTalker: Fine and Coarse Grained Context Modeling for Expressive Conversational Speech Synthesis","date":"2022-10-27","arxiv_id":"2210.15360","repositories_listed":1,"syntology":null},{"url":"/paper/empirical-study-incorporating-linguistic","slug":"empirical-study-incorporating-linguistic","title":"Empirical Study Incorporating Linguistic Knowledge on Filled Pauses for Personalized Spontaneous Speech Synthesis","date":"2022-10-14","arxiv_id":"2210.07559","repositories_listed":1,"syntology":null},{"url":"/paper/gan-you-hear-me-reclaiming-unconditional","slug":"gan-you-hear-me-reclaiming-unconditional","title":"GAN You Hear Me? Reclaiming Unconditional Speech Synthesis from Diffusion Models","date":"2022-10-11","arxiv_id":"2210.05271","repositories_listed":1,"syntology":null},{"url":"/paper/the-sound-of-silence-efficiency-of-first","slug":"the-sound-of-silence-efficiency-of-first","title":"The Sound of Silence: Efficiency of First Digit Features in Synthetic Audio Detection","date":"2022-10-06","arxiv_id":"2210.02746","repositories_listed":1,"syntology":null},{"url":"/paper/voice-spoofing-countermeasures-taxonomy-state","slug":"voice-spoofing-countermeasures-taxonomy-state","title":"Voice Spoofing Countermeasures: Taxonomy, State-of-the-art, experimental analysis of generalizability, open challenges, and the way forward","date":"2022-10-02","arxiv_id":"2210.00417","repositories_listed":1,"syntology":null},{"url":"/paper/detection-of-prosodic-boundaries-in-speech","slug":"detection-of-prosodic-boundaries-in-speech","title":"Detection of Prosodic Boundaries in Speech Using Wav2Vec 2.0","date":"2022-09-29","arxiv_id":"2209.15032","repositories_listed":1,"syntology":null},{"url":"/paper/controlvc-zero-shot-voice-conversion-with","slug":"controlvc-zero-shot-voice-conversion-with","title":"ControlVC: Zero-Shot Voice Conversion with Time-Varying Controls on Pitch and Speed","date":"2022-09-23","arxiv_id":"2209.11866","repositories_listed":1,"syntology":null},{"url":"/paper/mntts-an-open-source-mongolian-text-to-speech","slug":"mntts-an-open-source-mongolian-text-to-speech","title":"MnTTS: An Open-Source Mongolian Text-to-Speech Synthesis Dataset and Accompanied Baseline","date":"2022-09-22","arxiv_id":"2209.10848","repositories_listed":1,"syntology":null},{"url":"/paper/deep-speech-synthesis-from-articulatory-1","slug":"deep-speech-synthesis-from-articulatory-1","title":"Deep Speech Synthesis from Articulatory Representations","date":"2022-09-13","arxiv_id":"2209.06337","repositories_listed":1,"syntology":null},{"url":"/paper/mlphon-a-multifunctional-grapheme-phoneme","slug":"mlphon-a-multifunctional-grapheme-phoneme","title":"Mlphon: A Multifunctional Grapheme-Phoneme Conversion Tool Using Finite State Transducers","date":"2022-09-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/visualising-model-training-via-vowel-space","slug":"visualising-model-training-via-vowel-space","title":"Visualising Model Training via Vowel Space for Text-To-Speech Systems","date":"2022-08-21","arxiv_id":"2208.09775","repositories_listed":1,"syntology":null},{"url":"/paper/soundchoice-grapheme-to-phoneme-models-with","slug":"soundchoice-grapheme-to-phoneme-models-with","title":"SoundChoice: Grapheme-to-Phoneme Models with Semantic Disambiguation","date":"2022-07-27","arxiv_id":"2207.13703","repositories_listed":1,"syntology":null},{"url":"/paper/speaker-anonymization-with-phonetic","slug":"speaker-anonymization-with-phonetic","title":"Speaker Anonymization with Phonetic Intermediate Representations","date":"2022-07-11","arxiv_id":"2207.04834","repositories_listed":1,"syntology":null},{"url":"/paper/fastlts-non-autoregressive-end-to-end","slug":"fastlts-non-autoregressive-end-to-end","title":"FastLTS: Non-Autoregressive End-to-End Unconstrained Lip-to-Speech Synthesis","date":"2022-07-08","arxiv_id":"2207.03800","repositories_listed":1,"syntology":null},{"url":"/paper/building-african-voices","slug":"building-african-voices","title":"Building African Voices","date":"2022-07-01","arxiv_id":"2207.00688","repositories_listed":1,"syntology":null},{"url":"/paper/show-me-your-face-and-i-ll-tell-you-how-you","slug":"show-me-your-face-and-i-ll-tell-you-how-you","title":"Show Me Your Face, And I'll Tell You How You Speak","date":"2022-06-28","arxiv_id":"2206.14009","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-prosody-annotation-with-pre-trained","slug":"automatic-prosody-annotation-with-pre-trained","title":"Automatic Prosody Annotation with Pre-Trained Text-Speech Model","date":"2022-06-16","arxiv_id":"2206.07956","repositories_listed":1,"syntology":null},{"url":"/paper/styletts-a-style-based-generative-model-for","slug":"styletts-a-style-based-generative-model-for","title":"StyleTTS: A Style-Based Generative Model for Natural and Diverse Text-to-Speech Synthesis","date":"2022-05-30","arxiv_id":"2205.15439","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/styletts-a-style-based-generative-model-for#ran","syntology_url":"https://syntology.ai/paper/2205.15439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15439"}},"official":{"repos":["yl4579/StyleTTS"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transpeech-speech-to-speech-translation-with","slug":"transpeech-speech-to-speech-translation-with","title":"TranSpeech: Speech-to-Speech Translation With Bilateral Perturbation","date":"2022-05-25","arxiv_id":"2205.12523","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-zero-shot-voice-style-transfer","slug":"end-to-end-zero-shot-voice-style-transfer","title":"End-to-End Zero-Shot Voice Conversion with Location-Variable Convolutions","date":"2022-05-19","arxiv_id":"2205.09784","repositories_listed":1,"syntology":null},{"url":"/paper/sds-200-a-swiss-german-speech-to-standard","slug":"sds-200-a-swiss-german-speech-to-standard","title":"SDS-200: A Swiss German Speech to Standard German Text Corpus","date":"2022-05-19","arxiv_id":"2205.09501","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-packet-loss-concealment-with-mixed","slug":"real-time-packet-loss-concealment-with-mixed","title":"Real-Time Packet Loss Concealment With Mixed Generative and Predictive Model","date":"2022-05-11","arxiv_id":"2205.05785","repositories_listed":1,"syntology":null},{"url":"/paper/read-the-room-adapting-a-robot-s-voice-to","slug":"read-the-room-adapting-a-robot-s-voice-to","title":"Read the Room: Adapting a Robot's Voice to Ambient and Social Contexts","date":"2022-05-10","arxiv_id":"2205.04952","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-enabled-semantic-communications","slug":"deep-learning-enabled-semantic-communications","title":"Deep Learning Enabled Semantic Communications with Speech Recognition and Synthesis","date":"2022-05-09","arxiv_id":"2205.04603","repositories_listed":1,"syntology":null},{"url":"/paper/requirements-and-motivations-of-low-resource","slug":"requirements-and-motivations-of-low-resource","title":"Requirements and Motivations of Low-Resource Speech Synthesis for Language Revitalization","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/systematic-inequalities-in-language-1","slug":"systematic-inequalities-in-language-1","title":"Systematic Inequalities in Language Technology Performance across the World’s Languages","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-self-supervised-learning-based-mos","slug":"improving-self-supervised-learning-based-mos","title":"Improving Self-Supervised Learning-based MOS Prediction Networks","date":"2022-04-23","arxiv_id":"2204.11030","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-non-autoregressive-generation-for","slug":"a-survey-on-non-autoregressive-generation-for","title":"A Survey on Non-Autoregressive Generation for Neural Machine Translation and Beyond","date":"2022-04-20","arxiv_id":"2204.09269","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-strategies-for-articulatory","slug":"exploration-strategies-for-articulatory","title":"Exploration strategies for articulatory synthesis of complex syllable onsets","date":"2022-04-20","arxiv_id":"2204.09381","repositories_listed":1,"syntology":null},{"url":"/paper/lip-to-speech-synthesis-with-visual-context-1","slug":"lip-to-speech-synthesis-with-visual-context-1","title":"Lip to Speech Synthesis with Visual Context Attentional GAN","date":"2022-04-04","arxiv_id":"2204.01726","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lip-to-speech-synthesis-with-visual-context-1#ran","syntology_url":"https://syntology.ai/paper/2204.01726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01726"}},"official":{"repos":["ms-dot-k/Visual-Context-Attentional-GAN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-adaptor-converting-mel-spectrograms","slug":"universal-adaptor-converting-mel-spectrograms","title":"Universal Adaptor: Converting Mel-Spectrograms Between Different Configurations for Speech Synthesis","date":"2022-04-01","arxiv_id":"2204.00170","repositories_listed":1,"syntology":null},{"url":"/paper/a-single-speaker-is-almost-all-you-need-for","slug":"a-single-speaker-is-almost-all-you-need-for","title":"ASR data augmentation in low-resource settings using cross-lingual multi-speaker TTS and cross-lingual voice conversion","date":"2022-03-29","arxiv_id":"2204.00618","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-text-to-speech-synthesis-by","slug":"unsupervised-text-to-speech-synthesis-by","title":"Unsupervised Text-to-Speech Synthesis by Unsupervised Automatic Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15796","repositories_listed":1,"syntology":null},{"url":"/paper/bddm-bilateral-denoising-diffusion-models-for-1","slug":"bddm-bilateral-denoising-diffusion-models-for-1","title":"BDDM: Bilateral Denoising Diffusion Models for Fast and High-Quality Speech Synthesis","date":"2022-03-25","arxiv_id":"2203.13508","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bddm-bilateral-denoising-diffusion-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2203.13508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13508"}},"official":{"repos":["tencent-ailab/bddm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ecapa-tdnn-for-multi-speaker-text-to-speech","slug":"ecapa-tdnn-for-multi-speaker-text-to-speech","title":"ECAPA-TDNN for Multi-speaker Text-to-speech Synthesis","date":"2022-03-20","arxiv_id":"2203.10473","repositories_listed":1,"syntology":null},{"url":"/paper/generative-modeling-for-low-dimensional","slug":"generative-modeling-for-low-dimensional","title":"Generative Modeling for Low Dimensional Speech Attributes with Neural Spline Flows","date":"2022-03-03","arxiv_id":"2203.01786","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-lpcnet-a-neural-vocoder-with-fully","slug":"end-to-end-lpcnet-a-neural-vocoder-with-fully","title":"End-to-end LPCNet: A Neural Vocoder With Fully-Differentiable LPC Estimation","date":"2022-02-23","arxiv_id":"2202.11301","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-real-time-measure-of-the-perception","slug":"towards-a-real-time-measure-of-the-perception","title":"Towards a Real-time Measure of the Perception of Anthropomorphism in Human-robot Interaction","date":"2022-01-24","arxiv_id":"2201.09595","repositories_listed":1,"syntology":null},{"url":"/paper/a-practical-guide-to-logical-access-voice","slug":"a-practical-guide-to-logical-access-voice","title":"A Practical Guide to Logical Access Voice Presentation Attack Detection","date":"2022-01-10","arxiv_id":"2201.03321","repositories_listed":1,"syntology":null},{"url":"/paper/vocbench-a-neural-vocoder-benchmark-for","slug":"vocbench-a-neural-vocoder-benchmark-for","title":"VocBench: A Neural Vocoder Benchmark for Speech Synthesis","date":"2021-12-06","arxiv_id":"2112.03099","repositories_listed":1,"syntology":null},{"url":"/paper/meta-tts-meta-learning-for-few-shot-speaker","slug":"meta-tts-meta-learning-for-few-shot-speaker","title":"Meta-TTS: Meta-Learning for Few-Shot Speaker Adaptive Text-to-Speech","date":"2021-11-07","arxiv_id":"2111.04040","repositories_listed":1,"syntology":null},{"url":"/paper/cross-lingual-transfer-for-speech-processing","slug":"cross-lingual-transfer-for-speech-processing","title":"Cross-lingual Transfer for Speech Processing using Acoustic Language Similarity","date":"2021-11-02","arxiv_id":"2111.01326","repositories_listed":1,"syntology":null},{"url":"/paper/synthesizing-speech-from-intracranial-depth","slug":"synthesizing-speech-from-intracranial-depth","title":"Synthesizing Speech from Intracranial Depth Electrodes using an Encoder-Decoder Framework","date":"2021-11-02","arxiv_id":"2111.01457","repositories_listed":1,"syntology":null},{"url":"/paper/fairseq-s-2-a-scalable-and-integrable-speech-1","slug":"fairseq-s-2-a-scalable-and-integrable-speech-1","title":"fairseq Sˆ2: A Scalable and Integrable Speech Synthesis Toolkit","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/itacotron-2-transfering-english-speech","slug":"itacotron-2-transfering-english-speech","title":"ITAcotron 2: Transfering English Speech Synthesis Architectures and Speech Features to Italian","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fmfcc-a-a-challenging-mandarin-dataset-for","slug":"fmfcc-a-a-challenging-mandarin-dataset-for","title":"FMFCC-A: A Challenging Mandarin Dataset for Synthetic Speech Detection","date":"2021-10-18","arxiv_id":"2110.09441","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-style-control-in-transformer","slug":"fine-grained-style-control-in-transformer","title":"Fine-grained style control in Transformer-based Text-to-speech Synthesis","date":"2021-10-12","arxiv_id":"2110.06306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-style-control-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2110.06306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06306"}},"official":{"repos":["b04901014/FG-transformer-TTS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-lifelong-learning-of-multilingual","slug":"towards-lifelong-learning-of-multilingual","title":"Towards Lifelong Learning of Multilingual Text-To-Speech Synthesis","date":"2021-10-09","arxiv_id":"2110.04482","repositories_listed":1,"syntology":null},{"url":"/paper/cross-speaker-emotion-transfer-based-on","slug":"cross-speaker-emotion-transfer-based-on","title":"Cross-speaker Emotion Transfer Based on Speaker Condition Layer Normalization and Semi-Supervised Training in Text-To-Speech","date":"2021-10-08","arxiv_id":"2110.04153","repositories_listed":1,"syntology":null},{"url":"/paper/mixer-tts-non-autoregressive-fast-and-compact","slug":"mixer-tts-non-autoregressive-fast-and-compact","title":"Mixer-TTS: non-autoregressive, fast and compact text-to-speech model conditioned on language model embeddings","date":"2021-10-07","arxiv_id":"2110.03584","repositories_listed":1,"syntology":null},{"url":"/paper/strengthnet-deep-learning-based-emotion","slug":"strengthnet-deep-learning-based-emotion","title":"StrengthNet: Deep Learning-based Emotion Strength Assessment for Emotional Speech Synthesis","date":"2021-10-07","arxiv_id":"2110.03156","repositories_listed":1,"syntology":null},{"url":"/paper/editts-score-based-editing-for-controllable","slug":"editts-score-based-editing-for-controllable","title":"EdiTTS: Score-based Editing for Controllable Text-to-Speech","date":"2021-10-06","arxiv_id":"2110.02584","repositories_listed":1,"syntology":null},{"url":"/paper/integrated-speech-and-gesture-synthesis","slug":"integrated-speech-and-gesture-synthesis","title":"Integrated Speech and Gesture Synthesis","date":"2021-08-25","arxiv_id":"2108.11436","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-sound-generation-using-neural","slug":"conditional-sound-generation-using-neural","title":"Conditional Sound Generation Using Neural Discrete Time-Frequency Representation Learning","date":"2021-07-21","arxiv_id":"2107.09998","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-sound-generation-using-neural#ran","syntology_url":"https://syntology.ai/paper/2107.09998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.09998"}},"official":{"repos":["liuxubo717/sound_generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extending-text-to-speech-synthesis-with","slug":"extending-text-to-speech-synthesis-with","title":"Extending Text-to-Speech Synthesis with Articulatory Movement Prediction using Ultrasound Tongue Imaging","date":"2021-07-12","arxiv_id":"2107.05550","repositories_listed":1,"syntology":null},{"url":"/paper/speech-synthesis-from-text-and-ultrasound","slug":"speech-synthesis-from-text-and-ultrasound","title":"Speech Synthesis from Text and Ultrasound Tongue Image-based Articulatory Input","date":"2021-07-05","arxiv_id":"2107.02003","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-neural-speech-synthesis","slug":"a-survey-on-neural-speech-synthesis","title":"A Survey on Neural Speech Synthesis","date":"2021-06-29","arxiv_id":"2106.15561","repositories_listed":1,"syntology":null},{"url":"/paper/fastpitchformant-source-filter-based","slug":"fastpitchformant-source-filter-based","title":"FastPitchFormant: Source-filter based Decomposed Modeling for Speech Synthesis","date":"2021-06-29","arxiv_id":"2106.15123","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-the-knowledge-from-normalizing","slug":"distilling-the-knowledge-from-normalizing","title":"Distilling the Knowledge from Conditional Normalizing Flows","date":"2021-06-24","arxiv_id":"2106.12699","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/distilling-the-knowledge-from-normalizing#ran","syntology_url":"https://syntology.ai/paper/2106.12699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.12699"}},"official":{"repos":["yandex-research/distill-nf"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/priorgrad-improving-conditional-denoising","slug":"priorgrad-improving-conditional-denoising","title":"PriorGrad: Improving Conditional Denoising Diffusion Models with Data-Dependent Adaptive Prior","date":"2021-06-11","arxiv_id":"2106.06406","repositories_listed":1,"syntology":null},{"url":"/paper/nvc-net-end-to-end-adversarial-voice","slug":"nvc-net-end-to-end-adversarial-voice","title":"NVC-Net: End-to-End Adversarial Voice Conversion","date":"2021-06-02","arxiv_id":"2106.00992","repositories_listed":1,"syntology":null},{"url":"/paper/rad-tts-parallel-flow-based-tts-with-robust","slug":"rad-tts-parallel-flow-based-tts-with-robust","title":"RAD-TTS: Parallel Flow-Based TTS with Robust Alignment Learning and Diverse Synthesis","date":"2021-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/byakto-speech-real-time-long-speech-synthesis","slug":"byakto-speech-real-time-long-speech-synthesis","title":"Byakto Speech: Real-time long speech synthesis with convolutional neural network: Transfer learning from English to Bangla","date":"2021-05-31","arxiv_id":"2106.03937","repositories_listed":1,"syntology":null},{"url":"/paper/speaker-disentanglement-in-video-to-speech","slug":"speaker-disentanglement-in-video-to-speech","title":"Speaker disentanglement in video-to-speech conversion","date":"2021-05-20","arxiv_id":"2105.09652","repositories_listed":1,"syntology":null},{"url":"/paper/phrase-break-prediction-with-bidirectional","slug":"phrase-break-prediction-with-bidirectional","title":"Phrase break prediction with bidirectional encoder representations in Japanese text-to-speech synthesis","date":"2021-04-26","arxiv_id":"2104.12395","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-based-assessment-of-synthetic","slug":"deep-learning-based-assessment-of-synthetic","title":"Deep Learning Based Assessment of Synthetic Speech Naturalness","date":"2021-04-23","arxiv_id":"2104.11673","repositories_listed":1,"syntology":null},{"url":"/paper/kazakhtts-an-open-source-kazakh-text-to","slug":"kazakhtts-an-open-source-kazakh-text-to","title":"KazakhTTS: An Open-Source Kazakh Text-to-Speech Synthesis Dataset","date":"2021-04-17","arxiv_id":"2104.08459","repositories_listed":1,"syntology":null},{"url":"/paper/talknet-2-non-autoregressive-depth-wise","slug":"talknet-2-non-autoregressive-depth-wise","title":"TalkNet 2: Non-Autoregressive Depth-Wise Separable Convolutional Model for Speech Synthesis with Explicit Pitch and Duration Prediction","date":"2021-04-16","arxiv_id":"2104.08189","repositories_listed":1,"syntology":null},{"url":"/paper/half-truth-a-partially-fake-audio-detection","slug":"half-truth-a-partially-fake-audio-detection","title":"Half-Truth: A Partially Fake Audio Detection Dataset","date":"2021-04-08","arxiv_id":"2104.03617","repositories_listed":1,"syntology":null},{"url":"/paper/diff-tts-a-denoising-diffusion-model-for-text","slug":"diff-tts-a-denoising-diffusion-model-for-text","title":"Diff-TTS: A Denoising Diffusion Model for Text-to-Speech","date":"2021-04-03","arxiv_id":"2104.01409","repositories_listed":1,"syntology":null}],"record_sha256":"97ccf6b9977bae543cc7f273da392a9a2721b22d5e831effc032b48204b5aec2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}