{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech/papers/12","list_of":"/task/text-to-speech","task":"Text to Speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":15,"rows_per_page":100,"rows":[1101,1200],"of":1419,"counts":{"archive_papers_tagged":1419,"with_a_code_link":399,"where_syntology_ran_a_sample":108,"not_listed_spam_title":0,"listed":1419,"listed_where_code_ran":108,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech","prev":"/task/text-to-speech/papers/11","next":"/task/text-to-speech/papers/13","papers":[{"url":null,"slug":"emphasis-control-for-parallel-neural-tts","title":"Emphasis control for parallel neural TTS","date":"2021-10-06","arxiv_id":"2110.03012","repositories_listed":0,"syntology":null},{"url":null,"slug":"gantron-emotional-speech-synthesis-with","title":"GANtron: Emotional Speech Synthesis with Generative Adversarial Networks","date":"2021-10-06","arxiv_id":"2110.03390","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-prosody-modeling-and-control-in","title":"Hierarchical prosody modeling and control in non-autoregressive parallel neural TTS","date":"2021-10-06","arxiv_id":"2110.02952","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-tts-an-end-to-end-speech-synthesis","title":"Prosody-TTS: An end-to-end speech synthesis system with prosody control","date":"2021-10-06","arxiv_id":"2110.02854","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-equalization-unsupervised-learning-of","title":"Style Equalization: Unsupervised Learning of Controllable Generative Sequence Models","date":"2021-10-06","arxiv_id":"2110.02891","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-interplay-between-sparsity-naturalness","title":"On the Interplay Between Sparsity, Naturalness, Intelligibility, and Prosody in Speech Synthesis","date":"2021-10-04","arxiv_id":"2110.01147","repositories_listed":0,"syntology":null},{"url":"/paper/neural-speech-synthesis-in-german","slug":"neural-speech-synthesis-in-german","title":"Neural Speech Synthesis in German","date":"2021-10-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-speaker-embedding-and-post","title":"Incorporating speaker embedding and post-filter network for improving speaker similarity of personalized speech synthesis system","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-sequence-to-sequence-networks","title":"Conditioning Sequence-to-sequence Networks with Learned Activations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-text-to-speech-with-untranscribed","title":"Guided-TTS:Text-to-Speech with Untranscribed Speech","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"flowvocoder-a-small-footprint-neural-vocoder","title":"FlowVocoder: A small Footprint Neural Vocoder based Normalizing flow for Speech Synthesis","date":"2021-09-27","arxiv_id":"2109.13675","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-proposal-of-automatic-error-correction-in","title":"A Proposal of Automatic Error Correction in Text","date":"2021-09-24","arxiv_id":"2112.01846","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-latency-incremental-text-to-speech","title":"Low-Latency Incremental Text-to-Speech Synthesis with Distilled Context Prediction Network","date":"2021-09-22","arxiv_id":"2109.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-neural-speech-synthesis","title":"On-device neural speech synthesis","date":"2021-09-17","arxiv_id":"2109.08710","repositories_listed":0,"syntology":null},{"url":null,"slug":"referee-towards-reference-free-cross-speaker","title":"Referee: Towards reference-free cross-speaker style transfer with low-quality data for expressive speech synthesis","date":"2021-09-08","arxiv_id":"2109.03439","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-transformer-based-framework-for","title":"A Unified Transformer-based Framework for Duplex Text Normalization","date":"2021-08-23","arxiv_id":"2108.09889","repositories_listed":0,"syntology":null},{"url":null,"slug":"fighting-game-commentator-with-pitch-and","title":"Fighting Game Commentator with Pitch and Loudness Adjustment Utilizing Highlight Cues","date":"2021-08-18","arxiv_id":"2108.08112","repositories_listed":0,"syntology":null},{"url":null,"slug":"gc-tts-few-shot-speaker-adaptation-with","title":"GC-TTS: Few-shot Speaker Adaptation with Geometric Constraints","date":"2021-08-16","arxiv_id":"2108.06890","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-audio-quality-for-expressive-neural","title":"Enhancing audio quality for expressive Neural Text-to-Speech","date":"2021-08-13","arxiv_id":"2108.06270","repositories_listed":0,"syntology":null},{"url":null,"slug":"rw-resnet-a-novel-speech-anti-spoofing-model","title":"RW-Resnet: A Novel Speech Anti-Spoofing Model Using Raw Waveform","date":"2021-08-12","arxiv_id":"2108.05684","repositories_listed":0,"syntology":null},{"url":null,"slug":"anyonenet-synchronized-speech-and-talking","title":"AnyoneNet: Synchronized Speech and Talking Head Generation for Arbitrary Person","date":"2021-08-09","arxiv_id":"2108.04325","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-speech-enabled-fixed-phrase-translator-for","title":"A Speech-enabled Fixed-phrase Translator for Healthcare Accessibility","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-audio-synthesis-and-audio-visual","title":"A Survey on Audio Synthesis and Audio-Visual Multimodal Processing","date":"2021-08-01","arxiv_id":"2108.00443","repositories_listed":0,"syntology":null},{"url":null,"slug":"bts-back-transcription-for-speech-to-text","title":"BTS: Back TranScription for Speech-to-Text Post-Processor using Text-to-Speech-to-Text","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-speaker-style-transfer-with-prosody","title":"Cross-speaker Style Transfer with Prosody Bottleneck in Neural Speech Synthesis","date":"2021-07-27","arxiv_id":"2107.12562","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-einstein-experience-fast-text-to","title":"Digital Einstein Experience: Fast Text-to-Speech for Conversational AI","date":"2021-07-21","arxiv_id":"2107.10658","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-prosody-modeling-for-asr-tts-based-voice","title":"On Prosody Modeling for ASR+TTS based Voice Conversion","date":"2021-07-20","arxiv_id":"2107.09477","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-learning-with-dynamic-transformer","title":"Federated Learning with Dynamic Transformer for Text to Speech","date":"2021-07-09","arxiv_id":"2107.08795","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaspeech-3-adaptive-text-to-speech-for","title":"AdaSpeech 3: Adaptive Text to Speech for Spontaneous Style","date":"2021-07-06","arxiv_id":"2107.02530","repositories_listed":0,"syntology":null},{"url":null,"slug":"location-location-enhancing-the-evaluation-of","title":"Location, Location: Enhancing the Evaluation of Text-to-Speech Synthesis Using the Rapid Prosody Transcription Paradigm","date":"2021-07-06","arxiv_id":"2107.02527","repositories_listed":0,"syntology":null},{"url":null,"slug":"ganspeech-adversarial-training-for-high","title":"GANSpeech: Adversarial Training for High-Fidelity Multi-Speaker Speech Synthesis","date":"2021-06-29","arxiv_id":"2106.15153","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-context-aware-transformers-for","title":"Hierarchical Context-Aware Transformers for Non-Autoregressive Text to Speech","date":"2021-06-29","arxiv_id":"2106.15144","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-spectrogram-modelling-for-neural","title":"Multi-Scale Spectrogram Modelling for Neural Text-to-Speech","date":"2021-06-29","arxiv_id":"2106.15649","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-tts-with-explicit-duration","title":"Non-Autoregressive TTS with Explicit Duration Modelling for Low-Resource Highly Expressive Speech","date":"2021-06-24","arxiv_id":"2106.12896","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-native-english-lexicon-creation-for","title":"Non-native English lexicon creation for bilingual speech synthesis","date":"2021-06-21","arxiv_id":"2106.10870","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-speech-vocoding-for-text-to","title":"Advances in Speech Vocoding for Text-to-Speech with Continuous Parameters","date":"2021-06-19","arxiv_id":"2106.10481","repositories_listed":0,"syntology":null},{"url":"/paper/emovie-a-mandarin-emotion-speech-dataset-with","slug":"emovie-a-mandarin-emotion-speech-dataset-with","title":"EMOVIE: A Mandarin Emotion Speech Dataset with a Simple Emotional Text-to-Speech Model","date":"2021-06-17","arxiv_id":"2106.09317","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-expressiveness-of-neural","title":"Improving the expressiveness of neural vocoding with non-affine Normalizing Flows","date":"2021-06-16","arxiv_id":"2106.08649","repositories_listed":0,"syntology":null},{"url":null,"slug":"adept-a-dataset-for-evaluating-prosody","title":"ADEPT: A Dataset for Evaluating Prosody Transfer","date":"2021-06-15","arxiv_id":"2106.08321","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrl-p-temporal-control-of-prosodic-variation","title":"Ctrl-P: Temporal Control of Prosodic Variation for Speech Synthesis","date":"2021-06-15","arxiv_id":"2106.08352","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learned-conditional-prior-for-the-vae","title":"A learned conditional prior for the VAE acoustic space of a TTS system","date":"2021-06-14","arxiv_id":"2106.10229","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthasr-unlocking-synthetic-data-for-speech","title":"SynthASR: Unlocking Synthetic Data for Speech Recognition","date":"2021-06-14","arxiv_id":"2106.07803","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multi-speaker-tts-prosody-variance","title":"Improving multi-speaker TTS prosody variance with a residual encoder and normalizing flows","date":"2021-06-10","arxiv_id":"2106.05762","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-bert-embedding-for-improving-prosody","title":"Speech BERT Embedding For Improving Prosody in Neural TTS","date":"2021-06-08","arxiv_id":"2106.04312","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-methods-for-end-to-end","title":"Data Augmentation Methods for End-to-end Speech Recognition on Distant-Talk Scenarios","date":"2021-06-07","arxiv_id":"2106.03419","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforce-aligner-reinforcement-alignment","title":"Reinforce-Aligner: Reinforcement Alignment Search for Robust End-to-End Text-to-Speech","date":"2021-06-05","arxiv_id":"2106.02830","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-objective-evaluation-of-the-effects-of","title":"An objective evaluation of the effects of recording conditions and speaker characteristics in multi-speaker deep neural speech synthesis","date":"2021-06-03","arxiv_id":"2106.01812","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-verification-derived-loss-and-data","title":"Speaker verification-derived loss and data augmentation for DNN-based multispeaker speech synthesis","date":"2021-06-03","arxiv_id":"2106.01789","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-script-e2e-framework-for-multilingual","title":"Dual Script E2E framework for Multilingual and Code-Switching ASR","date":"2021-06-02","arxiv_id":"2106.01400","repositories_listed":0,"syntology":null},{"url":"/paper/a-corpus-of-neutral-voice-speech-in-brazilian","slug":"a-corpus-of-neutral-voice-speech-in-brazilian","title":"A Corpus of Neutral Voice Speech in Brazilian Portuguese","date":"2021-05-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robust-latent-representations-for","title":"Learning Robust Latent Representations for Controllable Speech Synthesis","date":"2021-05-10","arxiv_id":"2105.04458","repositories_listed":0,"syntology":null},{"url":null,"slug":"talromur-a-large-icelandic-tts-corpus","title":"Talrómur: A large Icelandic TTS corpus","date":"2021-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-addressing-practical-challenges-for-rnn","title":"On Addressing Practical Challenges for RNN-Transducer","date":"2021-04-27","arxiv_id":"2105.00858","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-parsing-based-semantic","title":"Enhancing Word-Level Semantic Representation via Dependency Structure for Expressive Text-to-Speech Synthesis","date":"2021-04-14","arxiv_id":"2104.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-sequence-to-sequence-voice","title":"Non-autoregressive sequence-to-sequence voice conversion","date":"2021-04-14","arxiv_id":"2104.06793","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-the-benefit-of-synthetic-training","title":"Comparing the Benefit of Synthetic Training Data for Various Automatic Speech Recognition Architectures","date":"2021-04-12","arxiv_id":"2104.05379","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-machine-speech-chain-for-domain","title":"Exploring Machine Speech Chain for Domain Adaptation and Few-Shot Speaker Adaptation","date":"2021-04-08","arxiv_id":"2104.03815","repositories_listed":0,"syntology":null},{"url":null,"slug":"flavored-tacotron-conditional-learning-for","title":"Flavored Tacotron: Conditional Learning for Prosodic-linguistic Features","date":"2021-04-08","arxiv_id":"2104.04050","repositories_listed":0,"syntology":null},{"url":null,"slug":"grapheme-to-phoneme-transformer-model-for","title":"Grapheme-to-Phoneme Transformer Model for Transfer Learning Dialects","date":"2021-04-08","arxiv_id":"2104.04091","repositories_listed":0,"syntology":null},{"url":null,"slug":"hi-fi-multi-speaker-english-tts-dataset","title":"Hi-Fi Multi-Speaker English TTS Dataset","date":"2021-04-03","arxiv_id":"2104.01497","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-emotional-text-to","title":"Reinforcement Learning for Emotional Text-to-Speech Synthesis with Improved Emotion Discriminability","date":"2021-04-03","arxiv_id":"2104.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-text-to-speech-using-style-tag","title":"Expressive Text-to-Speech using Style Tag","date":"2021-04-01","arxiv_id":"2104.00436","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-dctts-efficient-deep-convolutional-text","title":"Fast DCTTS: Efficient Deep Convolutional Text-to-Speech","date":"2021-04-01","arxiv_id":"2104.00624","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-rate-attention-architecture-for-fast","title":"Multi-rate attention architecture for fast streamable Text-to-speech spectrum modeling","date":"2021-04-01","arxiv_id":"2104.00705","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-speaker-adaptation-for-text-to","title":"Continual Speaker Adaptation for Text-to-Speech Synthesis","date":"2021-03-26","arxiv_id":"2103.14512","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-vocoder-multi-resolution-discriminator-is","title":"GAN Vocoder: Multi-Resolution Discriminator Is All You Need","date":"2021-03-09","arxiv_id":"2103.05236","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-neural-text-to-speech-model-utilizing","title":"A Neural Text-to-Speech Model Utilizing Broadcast Data Mixed with Background Music","date":"2021-03-04","arxiv_id":"2103.03049","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-architectures-to-extrapolate-emotional","title":"Model architectures to extrapolate emotional expressions in DNN-based text-to-speech","date":"2021-02-20","arxiv_id":"2102.10345","repositories_listed":0,"syntology":null},{"url":null,"slug":"alternate-endings-improving-prosody-for","title":"Alternate Endings: Improving Prosody for Incremental Neural TTS with Predicted Future Text Input","date":"2021-02-19","arxiv_id":"2102.09914","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiovisual-speech-synthesis-a-brief","title":"AudioVisual Speech Synthesis: A brief literature review","date":"2021-02-18","arxiv_id":"2103.03927","repositories_listed":0,"syntology":null},{"url":null,"slug":"vara-tts-non-autoregressive-text-to-speech","title":"VARA-TTS: Non-Autoregressive Text-to-Speech Synthesis based on Very Deep VAE with Residual Attention","date":"2021-02-12","arxiv_id":"2102.06431","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-cloning-a-multi-speaker-text-to-speech","title":"Voice Cloning: a Multi-Speaker Text-to-Speech Synthesis Approach based on Transfer Learning","date":"2021-02-10","arxiv_id":"2102.05630","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-natural-and-controllable-cross","title":"Towards Natural and Controllable Cross-Lingual Voice Conversion Based on Neural TTS Model and Phonetic Posteriorgram","date":"2021-02-03","arxiv_id":"2102.01991","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-neural-voice-cloning","title":"Expressive Neural Voice Cloning","date":"2021-01-30","arxiv_id":"2102.00151","repositories_listed":0,"syntology":null},{"url":null,"slug":"triple-m-a-practical-neural-text-to-speech","title":"Triple M: A Practical Text-to-speech Synthesis System With Multi-guidance Attention And Multi-band Multi-time LPCNet","date":"2021-01-30","arxiv_id":"2102.00247","repositories_listed":0,"syntology":null},{"url":null,"slug":"emocat-language-agnostic-emotional-voice","title":"EmoCat: Language-agnostic Emotional Voice Conversion","date":"2021-01-14","arxiv_id":"2101.05695","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-coherent-spontaneous-speech-and","title":"Generating coherent spontaneous speech and gesture from text","date":"2021-01-14","arxiv_id":"2101.05684","repositories_listed":0,"syntology":null},{"url":null,"slug":"whispered-and-lombard-neural-speech-synthesis","title":"Whispered and Lombard Neural Speech Synthesis","date":"2021-01-13","arxiv_id":"2101.05313","repositories_listed":0,"syntology":null},{"url":"/paper/joint-audio-visual-deepfake-detection","slug":"joint-audio-visual-deepfake-detection","title":"Joint Audio-Visual Deepfake Detection","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"detection-of-lexical-stress-errors-in-non","title":"Detection of Lexical Stress Errors in Non-Native (L2) English with Data Augmentation and Attention","date":"2020-12-29","arxiv_id":"2012.14788","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-text-to-speech-with-frame-level","title":"Denoising Text to Speech with Frame-Level Noise Modeling","date":"2020-12-17","arxiv_id":"2012.09547","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-wavenet-conditioned-on-vae-latent","title":"Parallel WaveNet conditioned on VAE latent vectors","date":"2020-12-17","arxiv_id":"2012.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"syntactic-representation-learning-for-neural","title":"Syntactic representation learning for neural network based TTS with syntactic parse tree traversal","date":"2020-12-13","arxiv_id":"2012.06971","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-previous-acoustic-context-to-improve","title":"Using previous acoustic context to improve Text-to-Speech synthesis","date":"2020-12-07","arxiv_id":"2012.03763","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphpb-graphical-representations-of-prosody","title":"GraphPB: Graphical Representations of Prosody Boundary in Speech Synthesis","date":"2020-12-03","arxiv_id":"2012.02626","repositories_listed":0,"syntology":null},{"url":null,"slug":"individually-amplified-text-to-speech","title":"Text-to-speech for the hearing impaired","date":"2020-12-03","arxiv_id":"2012.02174","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-smartcall-vietnamese-text-to","title":"Development of Smartcall Vietnamese Text-to-Speech for VLSP 2020","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-prosodic-phrasing-of-vietnamese","title":"Improving prosodic phrasing of Vietnamese text-to-speech systems","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vietnamese-text-to-speech-shared-task-vlsp","title":"Vietnamese Text-To-Speech Shared Task VLSP 2020: Remaining problems with state-of-the-art techniques","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrap-an-end-to-end-asr-system-by","title":"Bootstrap an end-to-end ASR system by multilingual training, transfer learning, text-to-text mapping and synthetic audio","date":"2020-11-25","arxiv_id":"2011.12696","repositories_listed":0,"syntology":null},{"url":null,"slug":"fbwave-efficient-and-scalable-neural-vocoders","title":"FBWave: Efficient and Scalable Neural Vocoders for Streaming Text-To-Speech on the Edge","date":"2020-11-25","arxiv_id":"2011.12985","repositories_listed":0,"syntology":null},{"url":null,"slug":"synth2aug-cross-domain-speaker-recognition","title":"Synth2Aug: Cross-domain speaker recognition with TTS synthesized speech","date":"2020-11-24","arxiv_id":"2011.11818","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-synthetic-audio-to-improve-the","title":"Using Synthetic Audio to Improve The Recognition of Out-Of-Vocabulary Words in End-To-End ASR Systems","date":"2020-11-23","arxiv_id":"2011.11564","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-shallow-fusion-for-rnn-t-personalization","title":"Deep Shallow Fusion for RNN-T Personalization","date":"2020-11-16","arxiv_id":"2011.07754","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-ipa-based-tacotron-for-data-efficient","title":"Using IPA-Based Tacotron for Data Efficient Cross-Lingual Speaker Adaptation and Pronunciation Enhancement","date":"2020-11-12","arxiv_id":"2011.06392","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-expressive-text-to-speech-using","title":"Low-resource expressive text-to-speech using data augmentation","date":"2020-11-11","arxiv_id":"2011.05707","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-speech-to-speech-translation","title":"Simultaneous Speech-to-Speech Translation System with Neural Incremental ASR, MT, and TTS","date":"2020-11-10","arxiv_id":"2011.04845","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-style-modelling-and-transfer-in","title":"Fine-grained Style Modeling, Transfer and Prediction in Text-to-Speech Synthesis via Phone-Level Content-Style Disentanglement","date":"2020-11-08","arxiv_id":"2011.03943","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-prosody-modelling-with-cross","title":"Improving Prosody Modelling with Cross-Utterance BERT Embeddings for End-to-end Speech Synthesis","date":"2020-11-06","arxiv_id":"2011.05161","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-images-for-asr-and-tts-through","title":"Augmenting Images for ASR and TTS through Single-loop and Dual-loop Multimodal Chain Framework","date":"2020-11-04","arxiv_id":"2011.02099","repositories_listed":0,"syntology":null}],"record_sha256":"4654aabfe422dc30e4a62152bad8dd07541af8f9296cd152a0391ef8b8e230f0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}