{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/9","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":13,"rows_per_page":100,"rows":[801,900],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/8","next":"/task/speech-synthesis/papers/10","papers":[{"url":null,"slug":"cross-lingual-low-resource-speaker-adaptation","title":"Cross-lingual Low Resource Speaker Adaptation Using Phonological Features","date":"2021-11-17","arxiv_id":"2111.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-streaming-speech-synthesis-with","title":"High Quality Streaming Speech Synthesis with Low, Sentence-Length-Independent Latency","date":"2021-11-17","arxiv_id":"2111.09052","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-speech-recognition-and-synthesis","title":"Modeling speech recognition and synthesis simultaneously: Encoding and decoding lexical and sublexical semantic information into speech with no access to speech data","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-for-low-resource-languages","title":"Speech Synthesis for Low Resource Languages using Transliteration Enabled Transfer Learning","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-prosody-for-unseen-texts-in-speech","title":"Improving Prosody for Unseen Texts in Speech Synthesis by Utilizing Linguistic Information and Noisy Data","date":"2021-11-15","arxiv_id":"2111.07549","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-evaluation-metrics-for-speech-to","title":"Assessing Evaluation Metrics for Speech-to-Speech Translation","date":"2021-10-26","arxiv_id":"2110.13877","repositories_listed":0,"syntology":null},{"url":null,"slug":"synt-utilizing-imperfect-synthetic-data-to","title":"Synt++: Utilizing Imperfect Synthetic Data to Improve Speech Recognition","date":"2021-10-21","arxiv_id":"2110.11479","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-simultaneous-speech-to-speech","title":"Direct Simultaneous Speech-to-Speech Translation with Variational Monotonic Multihead Attention","date":"2021-10-15","arxiv_id":"2110.08250","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-speech-synthesis-for-speech-to","title":"From Start to Finish: Latency Reduction Strategies for Incremental Speech Synthesis in Simultaneous Speech-to-Speech Translation","date":"2021-10-15","arxiv_id":"2110.08214","repositories_listed":0,"syntology":null},{"url":null,"slug":"singgan-generative-adversarial-network-for","title":"SingGAN: Generative Adversarial Network For High-Fidelity Singing Voice Generation","date":"2021-10-14","arxiv_id":"2110.07468","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepa-a-deep-neural-analyzer-for-speech-and","title":"DeepA: A Deep Neural Analyzer For Speech And Singing Vocoding","date":"2021-10-13","arxiv_id":"2110.06434","repositories_listed":0,"syntology":null},{"url":null,"slug":"laughnet-synthesizing-laughter-utterances","title":"LaughNet: synthesizing laughter utterances from waveform silhouettes and a single laughter example","date":"2021-10-11","arxiv_id":"2110.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-multiple-reference-audios-and-style","title":"Using multiple reference audios and style embedding constraints for speech synthesis","date":"2021-10-09","arxiv_id":"2110.04451","repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-aware-text-to-speech-synthesis","title":"Environment Aware Text-to-Speech Synthesis","date":"2021-10-08","arxiv_id":"2110.03887","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloning-one-s-voice-using-very-limited-data","title":"Cloning one's voice using very limited data in the wild","date":"2021-10-07","arxiv_id":"2110.03347","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualtts-tts-with-accurate-lip-speech","title":"VisualTTS: TTS with Accurate Lip-Speech Synchronization for Automatic Voice Over","date":"2021-10-07","arxiv_id":"2110.03342","repositories_listed":0,"syntology":null},{"url":null,"slug":"gantron-emotional-speech-synthesis-with","title":"GANtron: Emotional Speech Synthesis with Generative Adversarial Networks","date":"2021-10-06","arxiv_id":"2110.03390","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-tts-an-end-to-end-speech-synthesis","title":"Prosody-TTS: An end-to-end speech synthesis system with prosody control","date":"2021-10-06","arxiv_id":"2110.02854","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-interplay-between-sparsity-naturalness","title":"On the Interplay Between Sparsity, Naturalness, Intelligibility, and Prosody in Speech Synthesis","date":"2021-10-04","arxiv_id":"2110.01147","repositories_listed":0,"syntology":null},{"url":"/paper/neural-speech-synthesis-in-german","slug":"neural-speech-synthesis-in-german","title":"Neural Speech Synthesis in German","date":"2021-10-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-speaker-embedding-and-post","title":"Incorporating speaker embedding and post-filter network for improving speaker similarity of personalized speech synthesis system","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-sequence-to-sequence-networks","title":"Conditioning Sequence-to-sequence Networks with Learned Activations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-text-to-speech-with-untranscribed","title":"Guided-TTS:Text-to-Speech with Untranscribed Speech","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-mlp-a-simple-mlp-architecture-for","title":"Speech-MLP: a simple MLP architecture for speech processing","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synclr-a-synthesis-framework-for-contrastive","title":"SynCLR: A Synthesis Framework for Contrastive Learning of out-of-domain Speech Representations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"flowvocoder-a-small-footprint-neural-vocoder","title":"FlowVocoder: A small Footprint Neural Vocoder based Normalizing flow for Speech Synthesis","date":"2021-09-27","arxiv_id":"2109.13675","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-latency-incremental-text-to-speech","title":"Low-Latency Incremental Text-to-Speech Synthesis with Distilled Context Prediction Network","date":"2021-09-22","arxiv_id":"2109.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"hello-it-s-me-deep-learning-based-speech","title":"\"Hello, It's Me\": Deep Learning-based Speech Synthesis Attacks in the Real World","date":"2021-09-20","arxiv_id":"2109.09598","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-neural-speech-synthesis","title":"On-device neural speech synthesis","date":"2021-09-17","arxiv_id":"2109.08710","repositories_listed":0,"syntology":null},{"url":null,"slug":"referee-towards-reference-free-cross-speaker","title":"Referee: Towards reference-free cross-speaker style transfer with low-quality data for expressive speech synthesis","date":"2021-09-08","arxiv_id":"2109.03439","repositories_listed":0,"syntology":null},{"url":null,"slug":"physiological-physical-feature-fusion-for","title":"Physiological-Physical Feature Fusion for Automatic Voice Spoofing Detection","date":"2021-09-01","arxiv_id":"2109.00913","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-sequence-to-sequence-speech-synthesis","title":"Neural Sequence-to-Sequence Speech Synthesis Using a Hidden Semi-Markov Model Based Structured Attention Mechanism","date":"2021-08-31","arxiv_id":"2108.13985","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-attention-bidirectional-deep-learning","title":"Full Attention Bidirectional Deep Learning Structure for Single Channel Speech Enhancement","date":"2021-08-27","arxiv_id":"2108.12105","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-transformer-based-framework-for","title":"A Unified Transformer-based Framework for Duplex Text Normalization","date":"2021-08-23","arxiv_id":"2108.09889","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-audio-quality-for-expressive-neural","title":"Enhancing audio quality for expressive Neural Text-to-Speech","date":"2021-08-13","arxiv_id":"2108.06270","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-streamwise-gan-vocoder-for-wideband-speech","title":"A Streamwise GAN Vocoder for Wideband Speech Coding at Very Low Bit Rate","date":"2021-08-09","arxiv_id":"2108.04051","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-pronunciation-prediction-accuracy","title":"Improved pronunciation prediction accuracy using morphology","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-speaker-style-transfer-with-prosody","title":"Cross-speaker Style Transfer with Prosody Bottleneck in Neural Speech Synthesis","date":"2021-07-27","arxiv_id":"2107.12562","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-potential-of-lexical","title":"Exploring the Potential of Lexical Paraphrases for Mitigating Noise-Induced Comprehension Errors","date":"2021-07-18","arxiv_id":"2107.08337","repositories_listed":0,"syntology":null},{"url":null,"slug":"location-location-enhancing-the-evaluation-of","title":"Location, Location: Enhancing the Evaluation of Text-to-Speech Synthesis Using the Rapid Prosody Transcription Paradigm","date":"2021-07-06","arxiv_id":"2107.02527","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-objective-evaluation-framework-for","title":"An Objective Evaluation Framework for Pathological Speech Synthesis","date":"2021-07-01","arxiv_id":"2107.00308","repositories_listed":0,"syntology":null},{"url":null,"slug":"ganspeech-adversarial-training-for-high","title":"GANSpeech: Adversarial Training for High-Fidelity Multi-Speaker Speech Synthesis","date":"2021-06-29","arxiv_id":"2106.15153","repositories_listed":0,"syntology":null},{"url":null,"slug":"preliminary-study-on-using-vector","title":"Preliminary study on using vector quantization latent spaces for TTS/VC systems with consistent performance","date":"2021-06-25","arxiv_id":"2106.13479","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-context-aware-conversational","title":"Controllable Context-aware Conversational Speech Synthesis","date":"2021-06-21","arxiv_id":"2106.10828","repositories_listed":0,"syntology":null},{"url":null,"slug":"glow-wavegan-learning-speech-representations","title":"Glow-WaveGAN: Learning Speech Representations from GAN-based Variational Auto-Encoder For High Fidelity Flow-based Speech Synthesis","date":"2021-06-21","arxiv_id":"2106.10831","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-native-english-lexicon-creation-for","title":"Non-native English lexicon creation for bilingual speech synthesis","date":"2021-06-21","arxiv_id":"2106.10870","repositories_listed":0,"syntology":null},{"url":null,"slug":"unitts-residual-learning-of-unified-embedding","title":"UniTTS: Residual Learning of Unified Embedding Space for Speech Style Control","date":"2021-06-21","arxiv_id":"2106.11171","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-speech-vocoding-for-text-to","title":"Advances in Speech Vocoding for Text-to-Speech with Continuous Parameters","date":"2021-06-19","arxiv_id":"2106.10481","repositories_listed":0,"syntology":null},{"url":"/paper/emovie-a-mandarin-emotion-speech-dataset-with","slug":"emovie-a-mandarin-emotion-speech-dataset-with","title":"EMOVIE: A Mandarin Emotion Speech Dataset with a Simple Emotional Text-to-Speech Model","date":"2021-06-17","arxiv_id":"2106.09317","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-flow-based-neural-network-for-time-domain","title":"A Flow-Based Neural Network for Time Domain Speech Enhancement","date":"2021-06-16","arxiv_id":"2106.09008","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrl-p-temporal-control-of-prosodic-variation","title":"Ctrl-P: Temporal Control of Prosodic Variation for Speech Synthesis","date":"2021-06-15","arxiv_id":"2106.08352","repositories_listed":0,"syntology":null},{"url":null,"slug":"pathological-voice-adaptation-with","title":"Pathological voice adaptation with autoencoder-based voice conversion","date":"2021-06-15","arxiv_id":"2106.08427","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-wavelet-vocoder-based","title":"Continuous Wavelet Vocoder-based Decomposition of Parametric Speech Waveform Synthesis","date":"2021-06-12","arxiv_id":"2106.06863","repositories_listed":0,"syntology":null},{"url":null,"slug":"sprachsynthese-state-of-the-art-in-englischer","title":"Sprachsynthese -- State-of-the-Art in englischer und deutscher Sprache","date":"2021-06-11","arxiv_id":"2106.06230","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-efficiently-sample-from-diffusion","title":"Learning to Efficiently Sample from Diffusion Probabilistic Models","date":"2021-06-07","arxiv_id":"2106.03802","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathematical-vocoder-algorithm-modified","title":"Mathematical Vocoder Algorithm : Modified Spectral Inversion for Efficient Neural Speech Synthesis","date":"2021-06-06","arxiv_id":"2106.03167","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-objective-evaluation-of-the-effects-of","title":"An objective evaluation of the effects of recording conditions and speaker characteristics in multi-speaker deep neural speech synthesis","date":"2021-06-03","arxiv_id":"2106.01812","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-verification-derived-loss-and-data","title":"Speaker verification-derived loss and data augmentation for DNN-based multispeaker speech synthesis","date":"2021-06-03","arxiv_id":"2106.01789","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-script-e2e-framework-for-multilingual","title":"Dual Script E2E framework for Multilingual and Code-Switching ASR","date":"2021-06-02","arxiv_id":"2106.01400","repositories_listed":0,"syntology":null},{"url":"/paper/a-corpus-of-neutral-voice-speech-in-brazilian","slug":"a-corpus-of-neutral-voice-speech-in-brazilian","title":"A Corpus of Neutral Voice Speech in Brazilian Portuguese","date":"2021-05-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robust-latent-representations-for","title":"Learning Robust Latent Representations for Controllable Speech Synthesis","date":"2021-05-10","arxiv_id":"2105.04458","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-practical-lip-to-speech-conversion","title":"Towards a practical lip-to-speech conversion system using deep neural networks and mobile application frontend","date":"2021-04-29","arxiv_id":"2104.14467","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-video-to-speech-synthesis-using","title":"End-to-End Video-To-Speech Synthesis using Generative Adversarial Networks","date":"2021-04-27","arxiv_id":"2104.13332","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adaptive-learning-based-generative","title":"An Adaptive Learning based Generative Adversarial Network for One-To-One Voice Conversion","date":"2021-04-25","arxiv_id":"2104.12159","repositories_listed":0,"syntology":null},{"url":null,"slug":"review-of-end-to-end-speech-synthesis","title":"Review of end-to-end speech synthesis technology based on deep learning","date":"2021-04-20","arxiv_id":"2104.09995","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-parsing-based-semantic","title":"Enhancing Word-Level Semantic Representation via Dependency Structure for Expressive Text-to-Speech Synthesis","date":"2021-04-14","arxiv_id":"2104.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"flavored-tacotron-conditional-learning-for","title":"Flavored Tacotron: Conditional Learning for Prosodic-linguistic Features","date":"2021-04-08","arxiv_id":"2104.04050","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-multi-scale-style-control-for","title":"Towards Multi-Scale Style Control for Expressive Speech Synthesis","date":"2021-04-08","arxiv_id":"2104.03521","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-emotional-text-to","title":"Reinforcement Learning for Emotional Text-to-Speech Synthesis with Improved Emotion Discriminability","date":"2021-04-03","arxiv_id":"2104.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-speaker-adaptation-for-text-to","title":"Continual Speaker Adaptation for Text-to-Speech Synthesis","date":"2021-03-26","arxiv_id":"2103.14512","repositories_listed":0,"syntology":null},{"url":"/paper/swissdial-parallel-multidialectal-corpus-of","slug":"swissdial-parallel-multidialectal-corpus-of","title":"SwissDial: Parallel Multidialectal Corpus of Spoken Swiss German","date":"2021-03-21","arxiv_id":"2103.11401","repositories_listed":0,"syntology":null},{"url":null,"slug":"alternate-endings-improving-prosody-for","title":"Alternate Endings: Improving Prosody for Incremental Neural TTS with Predicted Future Text Input","date":"2021-02-19","arxiv_id":"2102.09914","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiovisual-speech-synthesis-a-brief","title":"AudioVisual Speech Synthesis: A brief literature review","date":"2021-02-18","arxiv_id":"2103.03927","repositories_listed":0,"syntology":null},{"url":null,"slug":"vara-tts-non-autoregressive-text-to-speech","title":"VARA-TTS: Non-Autoregressive Text-to-Speech Synthesis based on Very Deep VAE with Residual Attention","date":"2021-02-12","arxiv_id":"2102.06431","repositories_listed":0,"syntology":null},{"url":"/paper/asvspoof-2019-spoofing-countermeasures-for","slug":"asvspoof-2019-spoofing-countermeasures-for","title":"ASVspoof 2019: spoofing countermeasures for the detection of synthesized, converted and replayed speech","date":"2021-02-11","arxiv_id":"2102.05889","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-cloning-a-multi-speaker-text-to-speech","title":"Voice Cloning: a Multi-Speaker Text-to-Speech Synthesis Approach based on Transfer Learning","date":"2021-02-10","arxiv_id":"2102.05630","repositories_listed":0,"syntology":null},{"url":null,"slug":"generacion-de-voces-artificiales-infantiles","title":"Generacion de voces artificiales infantiles en castellano con acento costarricense","date":"2021-02-02","arxiv_id":"2102.01692","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-with-your-hands-using-continuous-hand","title":"SPEAK WITH YOUR HANDS Using Continuous Hand Gestures to control Articulatory Speech Synthesizer","date":"2021-02-02","arxiv_id":"2102.01640","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-neural-vocoding-with-parallel","title":"Universal Neural Vocoding with Parallel WaveNet","date":"2021-02-01","arxiv_id":"2102.01106","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-neural-voice-cloning","title":"Expressive Neural Voice Cloning","date":"2021-01-30","arxiv_id":"2102.00151","repositories_listed":0,"syntology":null},{"url":null,"slug":"triple-m-a-practical-neural-text-to-speech","title":"Triple M: A Practical Text-to-speech Synthesis System With Multi-guidance Attention And Multi-band Multi-time LPCNet","date":"2021-01-30","arxiv_id":"2102.00247","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-coherent-spontaneous-speech-and","title":"Generating coherent spontaneous speech and gesture from text","date":"2021-01-14","arxiv_id":"2101.05684","repositories_listed":0,"syntology":null},{"url":null,"slug":"whispered-and-lombard-neural-speech-synthesis","title":"Whispered and Lombard Neural Speech Synthesis","date":"2021-01-13","arxiv_id":"2101.05313","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-as-augmentation-for-low","title":"Speech Synthesis as Augmentation for Low-Resource ASR","date":"2020-12-23","arxiv_id":"2012.13004","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-wavenet-conditioned-on-vae-latent","title":"Parallel WaveNet conditioned on VAE latent vectors","date":"2020-12-17","arxiv_id":"2012.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-adaptive-normalization-driven-multi","title":"Few Shot Adaptive Normalization Driven Multi-Speaker Speech Synthesis","date":"2020-12-14","arxiv_id":"2012.07252","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-previous-acoustic-context-to-improve","title":"Using previous acoustic context to improve Text-to-Speech synthesis","date":"2020-12-07","arxiv_id":"2012.03763","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphpb-graphical-representations-of-prosody","title":"GraphPB: Graphical Representations of Prosody Boundary in Speech Synthesis","date":"2020-12-03","arxiv_id":"2012.02626","repositories_listed":0,"syntology":null},{"url":null,"slug":"german-arabic-speech-to-speech-translation","title":"German-Arabic Speech-to-Speech Translation for Psychiatric Diagnosis","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ji-yu-shen-du-xue-xi-zhi-zhong-wen-wen-zi","title":"基於深度學習之中文文字轉台語語音合成系統初步探討 (A Preliminary Study on Deep Learning-based Chinese Text to Taiwanese Speech Synthesis System)","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiment-analysis-for-emotional-speech","title":"Sentiment Analysis for Emotional Speech Synthesis in a News Dialogue System","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tal-a-synchronised-multi-speaker-corpus-of","title":"TaL: a synchronised multi-speaker corpus of ultrasound tongue imaging, audio, and lip videos","date":"2020-11-19","arxiv_id":"2011.09804","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretraining-strategies-waveform-model-choice","title":"Pretraining Strategies, Waveform Model Choice, and Acoustic Configurations for Multi-Speaker End-to-End Speech Synthesis","date":"2020-11-10","arxiv_id":"2011.04839","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-speech-to-speech-translation","title":"Simultaneous Speech-to-Speech Translation System with Neural Incremental ASR, MT, and TTS","date":"2020-11-10","arxiv_id":"2011.04845","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-gans-to-synthesise-minimum-training","title":"Using GANs to Synthesise Minimum Training Data for Deepfake Generation","date":"2020-11-10","arxiv_id":"2011.05421","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-style-modelling-and-transfer-in","title":"Fine-grained Style Modeling, Transfer and Prediction in Text-to-Speech Synthesis via Phone-Level Content-Style Disentanglement","date":"2020-11-08","arxiv_id":"2011.03943","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-prosody-modelling-with-cross","title":"Improving Prosody Modelling with Cross-Utterance BERT Embeddings for End-to-end Speech Synthesis","date":"2020-11-06","arxiv_id":"2011.05161","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-images-for-asr-and-tts-through","title":"Augmenting Images for ASR and TTS through Single-loop and Dual-loop Multimodal Chain Framework","date":"2020-11-04","arxiv_id":"2011.02099","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-machine-speech-chain-towards","title":"Incremental Machine Speech Chain Towards Enabling Listening while Speaking in Real-time","date":"2020-11-04","arxiv_id":"2011.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodic-representation-learning-and","title":"Prosodic Representation Learning and Contextual Sampling for Neural Text-to-Speech","date":"2020-11-04","arxiv_id":"2011.02252","repositories_listed":0,"syntology":null}],"record_sha256":"fb0c4dfda7e966d46ebaea247f7ad50ea405a6fa3c9f92a8727bfabcf70d3135","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}