{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-1/papers/4","list_of":"/task/text-to-speech-1","task":"text-to-speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":15,"rows_per_page":100,"rows":[301,400],"of":1413,"counts":{"archive_papers_tagged":1413,"with_a_code_link":395,"where_syntology_ran_a_sample":106,"not_listed_spam_title":0,"listed":1413,"listed_where_code_ran":106,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-1","prev":"/task/text-to-speech-1/papers/3","next":"/task/text-to-speech-1/papers/5","papers":[{"url":"/paper/generative-modeling-for-low-dimensional","slug":"generative-modeling-for-low-dimensional","title":"Generative Modeling for Low Dimensional Speech Attributes with Neural Spline Flows","date":"2022-03-03","arxiv_id":"2203.01786","repositories_listed":1,"syntology":null},{"url":"/paper/a-practical-guide-to-logical-access-voice","slug":"a-practical-guide-to-logical-access-voice","title":"A Practical Guide to Logical Access Voice Presentation Attack Detection","date":"2022-01-10","arxiv_id":"2201.03321","repositories_listed":1,"syntology":null},{"url":"/paper/a-wearable-sensor-vest-for-social-humanoid","slug":"a-wearable-sensor-vest-for-social-humanoid","title":"A wearable sensor vest for social humanoid robots with GPGPU, IoT, and modular software architecture","date":"2022-01-06","arxiv_id":"2201.02192","repositories_listed":1,"syntology":null},{"url":"/paper/more-than-words-in-the-wild-visually-driven","slug":"more-than-words-in-the-wild-visually-driven","title":"More than Words: In-the-Wild Visually-Driven Prosody for Text-to-Speech","date":"2021-11-19","arxiv_id":"2111.10139","repositories_listed":1,"syntology":null},{"url":"/paper/meta-tts-meta-learning-for-few-shot-speaker","slug":"meta-tts-meta-learning-for-few-shot-speaker","title":"Meta-TTS: Meta-Learning for Few-Shot Speaker Adaptive Text-to-Speech","date":"2021-11-07","arxiv_id":"2111.04040","repositories_listed":1,"syntology":null},{"url":"/paper/fairseq-s-2-a-scalable-and-integrable-speech-1","slug":"fairseq-s-2-a-scalable-and-integrable-speech-1","title":"fairseq Sˆ2: A Scalable and Integrable Speech Synthesis Toolkit","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fmfcc-a-a-challenging-mandarin-dataset-for","slug":"fmfcc-a-a-challenging-mandarin-dataset-for","title":"FMFCC-A: A Challenging Mandarin Dataset for Synthetic Speech Detection","date":"2021-10-18","arxiv_id":"2110.09441","repositories_listed":1,"syntology":null},{"url":"/paper/espnet2-tts-extending-the-edge-of-tts","slug":"espnet2-tts-extending-the-edge-of-tts","title":"ESPnet2-TTS: Extending the Edge of TTS Research","date":"2021-10-15","arxiv_id":"2110.07840","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espnet2-tts-extending-the-edge-of-tts#ran","syntology_url":"https://syntology.ai/paper/2110.07840","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07840"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-style-control-in-transformer","slug":"fine-grained-style-control-in-transformer","title":"Fine-grained style control in Transformer-based Text-to-speech Synthesis","date":"2021-10-12","arxiv_id":"2110.06306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-style-control-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2110.06306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06306"}},"official":{"repos":["b04901014/FG-transformer-TTS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-lifelong-learning-of-multilingual","slug":"towards-lifelong-learning-of-multilingual","title":"Towards Lifelong Learning of Multilingual Text-To-Speech Synthesis","date":"2021-10-09","arxiv_id":"2110.04482","repositories_listed":1,"syntology":null},{"url":"/paper/cross-speaker-emotion-transfer-based-on","slug":"cross-speaker-emotion-transfer-based-on","title":"Cross-speaker Emotion Transfer Based on Speaker Condition Layer Normalization and Semi-Supervised Training in Text-To-Speech","date":"2021-10-08","arxiv_id":"2110.04153","repositories_listed":1,"syntology":null},{"url":"/paper/applying-phonological-features-in","slug":"applying-phonological-features-in","title":"Applying Phonological Features in Multilingual Text-To-Speech","date":"2021-10-07","arxiv_id":"2110.03609","repositories_listed":1,"syntology":null},{"url":"/paper/mixer-tts-non-autoregressive-fast-and-compact","slug":"mixer-tts-non-autoregressive-fast-and-compact","title":"Mixer-TTS: non-autoregressive, fast and compact text-to-speech model conditioned on language model embeddings","date":"2021-10-07","arxiv_id":"2110.03584","repositories_listed":1,"syntology":null},{"url":"/paper/editts-score-based-editing-for-controllable","slug":"editts-score-based-editing-for-controllable","title":"EdiTTS: Score-based Editing for Controllable Text-to-Speech","date":"2021-10-06","arxiv_id":"2110.02584","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-text-to-speech-for-text-based","slug":"zero-shot-text-to-speech-for-text-based","title":"Zero-Shot Text-to-Speech for Text-Based Insertion in Audio Narration","date":"2021-09-12","arxiv_id":"2109.05426","repositories_listed":1,"syntology":null},{"url":"/paper/integrated-speech-and-gesture-synthesis","slug":"integrated-speech-and-gesture-synthesis","title":"Integrated Speech and Gesture Synthesis","date":"2021-08-25","arxiv_id":"2108.11436","repositories_listed":1,"syntology":null},{"url":"/paper/adaptation-of-tacotron2-based-text-to-speech","slug":"adaptation-of-tacotron2-based-text-to-speech","title":"Adaptation of Tacotron2-based Text-To-Speech for Articulatory-to-Acoustic Mapping using Ultrasound Tongue Imaging","date":"2021-07-26","arxiv_id":"2107.12051","repositories_listed":1,"syntology":null},{"url":"/paper/extending-text-to-speech-synthesis-with","slug":"extending-text-to-speech-synthesis-with","title":"Extending Text-to-Speech Synthesis with Articulatory Movement Prediction using Ultrasound Tongue Imaging","date":"2021-07-12","arxiv_id":"2107.05550","repositories_listed":1,"syntology":null},{"url":"/paper/speech-synthesis-from-text-and-ultrasound","slug":"speech-synthesis-from-text-and-ultrasound","title":"Speech Synthesis from Text and Ultrasound Tongue Image-based Articulatory Input","date":"2021-07-05","arxiv_id":"2107.02003","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-neural-speech-synthesis","slug":"a-survey-on-neural-speech-synthesis","title":"A Survey on Neural Speech Synthesis","date":"2021-06-29","arxiv_id":"2106.15561","repositories_listed":1,"syntology":null},{"url":"/paper/fastpitchformant-source-filter-based","slug":"fastpitchformant-source-filter-based","title":"FastPitchFormant: Source-filter based Decomposed Modeling for Speech Synthesis","date":"2021-06-29","arxiv_id":"2106.15123","repositories_listed":1,"syntology":null},{"url":"/paper/hui-audio-corpus-german-a-high-quality-tts","slug":"hui-audio-corpus-german-a-high-quality-tts","title":"HUI-Audio-Corpus-German: A high quality TTS dataset","date":"2021-06-11","arxiv_id":"2106.06309","repositories_listed":1,"syntology":null},{"url":"/paper/wav2kws-transfer-learning-from-speech","slug":"wav2kws-transfer-learning-from-speech","title":"Wav2KWS: Transfer Learning from Speech Representations for Keyword Spotting","date":"2021-05-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/phrase-break-prediction-with-bidirectional","slug":"phrase-break-prediction-with-bidirectional","title":"Phrase break prediction with bidirectional encoder representations in Japanese text-to-speech synthesis","date":"2021-04-26","arxiv_id":"2104.12395","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-based-assessment-of-synthetic","slug":"deep-learning-based-assessment-of-synthetic","title":"Deep Learning Based Assessment of Synthetic Speech Naturalness","date":"2021-04-23","arxiv_id":"2104.11673","repositories_listed":1,"syntology":null},{"url":"/paper/adaspeech-2-adaptive-text-to-speech-with","slug":"adaspeech-2-adaptive-text-to-speech-with","title":"AdaSpeech 2: Adaptive Text to Speech with Untranscribed Data","date":"2021-04-20","arxiv_id":"2104.09715","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaspeech-2-adaptive-text-to-speech-with#ran","syntology_url":"https://syntology.ai/paper/2104.09715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09715"}},"official":null}},{"url":"/paper/kazakhtts-an-open-source-kazakh-text-to","slug":"kazakhtts-an-open-source-kazakh-text-to","title":"KazakhTTS: An Open-Source Kazakh Text-to-Speech Synthesis Dataset","date":"2021-04-17","arxiv_id":"2104.08459","repositories_listed":1,"syntology":null},{"url":"/paper/talknet-2-non-autoregressive-depth-wise","slug":"talknet-2-non-autoregressive-depth-wise","title":"TalkNet 2: Non-Autoregressive Depth-Wise Separable Convolutional Model for Speech Synthesis with Explicit Pitch and Duration Prediction","date":"2021-04-16","arxiv_id":"2104.08189","repositories_listed":1,"syntology":null},{"url":"/paper/proteno-text-normalization-with-limited-data","slug":"proteno-text-normalization-with-limited-data","title":"Proteno: Text Normalization with Limited Data for Fast Deployment in Text to Speech Systems","date":"2021-04-15","arxiv_id":"2104.07777","repositories_listed":1,"syntology":null},{"url":"/paper/nemo-toolbox-for-speech-dataset-construction","slug":"nemo-toolbox-for-speech-dataset-construction","title":"A Toolbox for Construction and Analysis of Speech Datasets","date":"2021-04-11","arxiv_id":"2104.04896","repositories_listed":1,"syntology":null},{"url":"/paper/ai4d-african-language-program","slug":"ai4d-african-language-program","title":"AI4D -- African Language Program","date":"2021-04-06","arxiv_id":"2104.02516","repositories_listed":1,"syntology":null},{"url":"/paper/diff-tts-a-denoising-diffusion-model-for-text","slug":"diff-tts-a-denoising-diffusion-model-for-text","title":"Diff-TTS: A Denoising Diffusion Model for Text-to-Speech","date":"2021-04-03","arxiv_id":"2104.01409","repositories_listed":1,"syntology":null},{"url":"/paper/attention-forcing-for-machine-translation","slug":"attention-forcing-for-machine-translation","title":"Attention Forcing for Machine Translation","date":"2021-04-02","arxiv_id":"2104.01264","repositories_listed":1,"syntology":null},{"url":"/paper/styler-style-modeling-with-rapidity-and","slug":"styler-style-modeling-with-rapidity-and","title":"STYLER: Style Factor Modeling with Rapidity and Robustness via Speech Decomposition for Expressive and Controllable Neural Text to Speech","date":"2021-03-17","arxiv_id":"2103.09474","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-on-incorporating-pretrained-and","slug":"investigating-on-incorporating-pretrained-and","title":"Investigating on Incorporating Pretrained and Learnable Speaker Representations for Multi-Speaker Multi-Style Text-to-Speech","date":"2021-03-06","arxiv_id":"2103.04088","repositories_listed":1,"syntology":null},{"url":"/paper/bidirectional-variational-inference-for-non","slug":"bidirectional-variational-inference-for-non","title":"Bidirectional Variational Inference for Non-Autoregressive Text-to-Speech","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unified-mandarin-tts-front-end-based-on","slug":"unified-mandarin-tts-front-end-based-on","title":"Unified Mandarin TTS Front-end Based on Distilled BERT Model","date":"2020-12-31","arxiv_id":"2012.15404","repositories_listed":1,"syntology":null},{"url":"/paper/mls-a-large-scale-multilingual-dataset-for","slug":"mls-a-large-scale-multilingual-dataset-for","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","date":"2020-12-07","arxiv_id":"2012.03411","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-generalization-learning-in-low","slug":"cross-modal-generalization-learning-in-low","title":"Cross-Modal Generalization: Learning in Low Resource Modalities via Meta-Alignment","date":"2020-12-04","arxiv_id":"2012.02813","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-url-segmentation-with-1","slug":"semi-supervised-url-segmentation-with-1","title":"Semi-supervised URL Segmentation with Recurrent Neural Networks Pre-trained on Knowledge Graph Entities","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/empirical-evaluation-of-deep-learning-model","slug":"empirical-evaluation-of-deep-learning-model","title":"Empirical Evaluation of Deep Learning Model Compression Techniques on the WaveNet Vocoder","date":"2020-11-20","arxiv_id":"2011.10469","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-prosody-modeling-for-non","slug":"hierarchical-prosody-modeling-for-non","title":"Hierarchical Prosody Modeling for Non-Autoregressive Speech Synthesis","date":"2020-11-12","arxiv_id":"2011.06465","repositories_listed":1,"syntology":null},{"url":"/paper/naturalization-of-text-by-the-insertion-of","slug":"naturalization-of-text-by-the-insertion-of","title":"Naturalization of Text by the Insertion of Pauses and Filler Words","date":"2020-11-07","arxiv_id":"2011.03713","repositories_listed":1,"syntology":null},{"url":"/paper/wave-tacotron-spectrogram-free-end-to-end","slug":"wave-tacotron-spectrogram-free-end-to-end","title":"Wave-Tacotron: Spectrogram-free end-to-end text-to-speech synthesis","date":"2020-11-06","arxiv_id":"2011.03568","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-url-segmentation-with","slug":"semi-supervised-url-segmentation-with","title":"Semi-supervised URL Segmentation with Recurrent Neural NetworksPre-trained on Knowledge Graph Entities","date":"2020-11-05","arxiv_id":"2011.03138","repositories_listed":1,"syntology":null},{"url":"/paper/effective-deep-learning-models-for-automatic","slug":"effective-deep-learning-models-for-automatic","title":"Effective Deep Learning Models for Automatic Diacritization of Arabic Text","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/iestac-english-italian-parallel-corpus-for","slug":"iestac-english-italian-parallel-corpus-for","title":"IESTAC: English-Italian Parallel Corpus for End-to-End Speech-to-Text Machine Translation","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-speaker-embedding-from-text-to","slug":"learning-speaker-embedding-from-text-to","title":"Learning Speaker Embedding from Text-to-Speech","date":"2020-10-21","arxiv_id":"2010.11221","repositories_listed":1,"syntology":null},{"url":"/paper/towards-natural-bilingual-and-code-switched","slug":"towards-natural-bilingual-and-code-switched","title":"Towards Natural Bilingual and Code-Switched Speech Synthesis Based on Mix of Monolingual Recordings and Cross-Lingual Voice Conversion","date":"2020-10-16","arxiv_id":"2010.08136","repositories_listed":1,"syntology":null},{"url":"/paper/google-crowdsourced-speech-corpora-and","slug":"google-crowdsourced-speech-corpora-and","title":"Google Crowdsourced Speech Corpora and Related Open-Source Resources for Low-Resource Languages and Dialects: An Overview","date":"2020-10-14","arxiv_id":"2010.06778","repositories_listed":1,"syntology":null},{"url":"/paper/jsss-free-japanese-speech-corpus-for","slug":"jsss-free-japanese-speech-corpus-for","title":"JSSS: free Japanese speech corpus for summarization and simplification","date":"2020-10-05","arxiv_id":"2010.01793","repositories_listed":1,"syntology":null},{"url":"/paper/accent-estimation-of-japanese-words-from","slug":"accent-estimation-of-japanese-words-from","title":"Accent Estimation of Japanese Words from Their Surfaces and Romanizations for Building Large Vocabulary Accent Dictionaries","date":"2020-09-21","arxiv_id":"2009.09679","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-speech-intelligibility-in-text-to","slug":"enhancing-speech-intelligibility-in-text-to","title":"Enhancing Speech Intelligibility in Text-To-Speech Synthesis using Speaking Style Conversion","date":"2020-08-13","arxiv_id":"2008.05809","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-speech-intelligibility-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2008.05809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05809"}},"official":null}},{"url":"/paper/attentron-few-shot-text-to-speech-utilizing-1","slug":"attentron-few-shot-text-to-speech-utilizing-1","title":"Attentron: Few-Shot Text-to-Speech Utilizing Attention-Based Variable-Length Embedding","date":"2020-08-12","arxiv_id":"2005.08484","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/attentron-few-shot-text-to-speech-utilizing-1#ran","syntology_url":"https://syntology.ai/paper/2005.08484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08484"}},"official":null}},{"url":"/paper/speaker-conditional-wavernn-towards-universal","slug":"speaker-conditional-wavernn-towards-universal","title":"Speaker Conditional WaveRNN: Towards Universal Neural Vocoder for Unseen Speaker and Recording Conditions","date":"2020-08-09","arxiv_id":"2008.05289","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speaker-conditional-wavernn-towards-universal#ran","syntology_url":"https://syntology.ai/paper/2008.05289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05289"}},"official":null}},{"url":"/paper/phonological-features-for-0-shot-multilingual","slug":"phonological-features-for-0-shot-multilingual","title":"Phonological Features for 0-shot Multilingual Speech Synthesis","date":"2020-08-06","arxiv_id":"2008.04107","repositories_listed":1,"syntology":null},{"url":"/paper/one-model-many-languages-meta-learning-for","slug":"one-model-many-languages-meta-learning-for","title":"One Model, Many Languages: Meta-learning for Multilingual Text-to-Speech","date":"2020-08-03","arxiv_id":"2008.00768","repositories_listed":1,"syntology":null},{"url":"/paper/multispeech-multi-speaker-text-to-speech-with","slug":"multispeech-multi-speaker-text-to-speech-with","title":"MultiSpeech: Multi-Speaker Text to Speech with Transformer","date":"2020-06-08","arxiv_id":"2006.04664","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multispeech-multi-speaker-text-to-speech-with#ran","syntology_url":"https://syntology.ai/paper/2006.04664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04664"}},"official":null}},{"url":"/paper/exploring-tts-without-t-using-biologically","slug":"exploring-tts-without-t-using-biologically","title":"Exploring TTS without T Using Biologically/Psychologically Motivated Neural Network Modules (ZeroSpeech 2020)","date":"2020-05-11","arxiv_id":"2005.05487","repositories_listed":1,"syntology":null},{"url":"/paper/luganda-text-to-speech-machine","slug":"luganda-text-to-speech-machine","title":"Luganda Text-to-Speech Machine","date":"2020-05-11","arxiv_id":"2005.05447","repositories_listed":1,"syntology":null},{"url":"/paper/from-speaker-verification-to-multispeaker","slug":"from-speaker-verification-to-multispeaker","title":"From Speaker Verification to Multispeaker Speech Synthesis, Deep Transfer with Feedback Constraint","date":"2020-05-10","arxiv_id":"2005.04587","repositories_listed":1,"syntology":null},{"url":"/paper/espnet-st-all-in-one-speech-translation","slug":"espnet-st-all-in-one-speech-translation","title":"ESPnet-ST: All-in-One Speech Translation Toolkit","date":"2020-04-21","arxiv_id":"2004.10234","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-grapheme-to-phoneme-1","slug":"transformer-based-grapheme-to-phoneme-1","title":"Transformer based Grapheme-to-Phoneme Conversion","date":"2020-04-14","arxiv_id":"2004.06338","repositories_listed":1,"syntology":null},{"url":"/paper/g2pm-a-neural-grapheme-to-phoneme-conversion","slug":"g2pm-a-neural-grapheme-to-phoneme-conversion","title":"g2pM: A Neural Grapheme-to-Phoneme Conversion Package for Mandarin Chinese Based on a New Open Benchmark Dataset","date":"2020-04-07","arxiv_id":"2004.03136","repositories_listed":1,"syntology":null},{"url":"/paper/perception-of-prosodic-variation-for-speech","slug":"perception-of-prosodic-variation-for-speech","title":"Perception of prosodic variation for speech synthesis using an unsupervised discrete representation of F0","date":"2020-03-14","arxiv_id":"2003.06686","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-discrepancy-between-density-estimation","slug":"on-the-discrepancy-between-density-estimation","title":"On the Discrepancy between Density Estimation and Sequence Generation","date":"2020-02-17","arxiv_id":"2002.07233","repositories_listed":1,"syntology":null},{"url":"/paper/improving-lpcnet-based-text-to-speech-with","slug":"improving-lpcnet-based-text-to-speech-with","title":"Improving LPCNet-based Text-to-Speech with Linear Prediction-structured Mixture Density Network","date":"2020-01-31","arxiv_id":"2001.11686","repositories_listed":1,"syntology":null},{"url":"/paper/generating-synthetic-audio-data-for-attention","slug":"generating-synthetic-audio-data-for-attention","title":"Generating Synthetic Audio Data for Attention-Based Speech Recognition Systems","date":"2019-12-19","arxiv_id":"1912.09257","repositories_listed":1,"syntology":null},{"url":"/paper/neural-voice-puppetry-audio-driven-facial","slug":"neural-voice-puppetry-audio-driven-facial","title":"Neural Voice Puppetry: Audio-driven Facial Reenactment","date":"2019-12-11","arxiv_id":"1912.05566","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-mask-for-transformer-based-end-to","slug":"semantic-mask-for-transformer-based-end-to","title":"Semantic Mask for Transformer based End-to-End Speech Recognition","date":"2019-12-06","arxiv_id":"1912.03010","repositories_listed":1,"syntology":null},{"url":"/paper/independent-and-automatic-evaluation-of","slug":"independent-and-automatic-evaluation-of","title":"Independent and automatic evaluation of acoustic-to-articulatory inversion models","date":"2019-11-15","arxiv_id":"1911.06573","repositories_listed":1,"syntology":null},{"url":"/paper/emotional-voice-conversion-using-multitask","slug":"emotional-voice-conversion-using-multitask","title":"Emotional Voice Conversion using Multitask Learning with Text-to-speech","date":"2019-11-11","arxiv_id":"1911.06149","repositories_listed":1,"syntology":null},{"url":"/paper/spoofing-speaker-verification-systems-with","slug":"spoofing-speaker-verification-systems-with","title":"Spoofing Speaker Verification Systems with Deep Multi-speaker Text-to-speech Synthesis","date":"2019-10-29","arxiv_id":"1910.13054","repositories_listed":1,"syntology":null},{"url":"/paper/numbers-normalisation-in-the-inflected","slug":"numbers-normalisation-in-the-inflected","title":"Numbers Normalisation in the Inflected Languages: a Case Study of Polish","date":"2019-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mass-a-large-and-clean-multilingual-corpus-of","slug":"mass-a-large-and-clean-multilingual-corpus-of","title":"MaSS: A Large and Clean Multilingual Corpus of Sentence-aligned Spoken Utterances Extracted from the Bible","date":"2019-07-30","arxiv_id":"1907.12895","repositories_listed":1,"syntology":null},{"url":"/paper/attention-model-for-articulatory-features","slug":"attention-model-for-articulatory-features","title":"Attention model for articulatory features detection","date":"2019-07-02","arxiv_id":"1907.01914","repositories_listed":1,"syntology":null},{"url":"/paper/using-generative-modelling-to-produce-varied","slug":"using-generative-modelling-to-produce-varied","title":"Using generative modelling to produce varied intonation for speech synthesis","date":"2019-06-10","arxiv_id":"1906.04233","repositories_listed":1,"syntology":null},{"url":"/paper/effective-parameter-estimation-methods-for-an","slug":"effective-parameter-estimation-methods-for-an","title":"Effective parameter estimation methods for an ExcitNet model in generative text-to-speech systems","date":"2019-05-21","arxiv_id":"1905.08486","repositories_listed":1,"syntology":null},{"url":"/paper/expediting-tts-synthesis-with-adversarial","slug":"expediting-tts-synthesis-with-adversarial","title":"Expediting TTS Synthesis with Adversarial Vocoding","date":"2019-04-16","arxiv_id":"1904.07944","repositories_listed":1,"syntology":null},{"url":"/paper/direct-speech-to-speech-translation-with-a","slug":"direct-speech-to-speech-translation-with-a","title":"Direct speech-to-speech translation with a sequence-to-sequence model","date":"2019-04-12","arxiv_id":"1904.06037","repositories_listed":1,"syntology":null},{"url":"/paper/gelp-gan-excited-liner-prediction-for-speech","slug":"gelp-gan-excited-liner-prediction-for-speech","title":"GELP: GAN-Excited Linear Prediction for Speech Synthesis from Mel-spectrogram","date":"2019-04-08","arxiv_id":"1904.03976","repositories_listed":1,"syntology":null},{"url":"/paper/in-other-news-a-bi-style-text-to-speech-model","slug":"in-other-news-a-bi-style-text-to-speech-model","title":"In Other News: A Bi-style Text-to-speech Model for Synthesizing Newscaster Voice with Limited Data","date":"2019-04-04","arxiv_id":"1904.02790","repositories_listed":1,"syntology":null},{"url":"/paper/assert-anti-spoofing-with-squeeze-excitation","slug":"assert-anti-spoofing-with-squeeze-excitation","title":"ASSERT: Anti-Spoofing with Squeeze-Excitation and Residual neTworks","date":"2019-04-01","arxiv_id":"1904.01120","repositories_listed":1,"syntology":null},{"url":"/paper/css10-a-collection-of-single-speaker-speech","slug":"css10-a-collection-of-single-speaker-speech","title":"CSS10: A Collection of Single Speaker Speech Datasets for 10 Languages","date":"2019-03-27","arxiv_id":"1903.11269","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/css10-a-collection-of-single-speaker-speech#ran","syntology_url":"https://syntology.ai/paper/1903.11269","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.11269"}},"official":{"repos":["Kyubyong/css10"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/visualization-and-interpretation-of-latent","slug":"visualization-and-interpretation-of-latent","title":"Visualization and Interpretation of Latent Spaces for Controlling Expressive Speech Synthesis through Audio Analysis","date":"2019-03-27","arxiv_id":"1903.11570","repositories_listed":1,"syntology":null},{"url":"/paper/investigation-of-enhanced-tacotron-text-to","slug":"investigation-of-enhanced-tacotron-text-to","title":"Investigation of enhanced Tacotron text-to-speech synthesis systems with self-attention for pitch accent language","date":"2018-10-29","arxiv_id":"1810.11960","repositories_listed":1,"syntology":null},{"url":"/paper/a-fully-time-domain-neural-model-for-subband","slug":"a-fully-time-domain-neural-model-for-subband","title":"A Fully Time-domain Neural Model for Subband-based Speech Synthesizer","date":"2018-10-12","arxiv_id":"1810.05319","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-sequence-to-sequence-learning-for","slug":"attentive-sequence-to-sequence-learning-for","title":"Attentive Sequence-to-Sequence Learning for Diacritic Restoration of Yorùbá Language Text","date":"2018-04-03","arxiv_id":"1804.00832","repositories_listed":1,"syntology":null},{"url":"/paper/obamanet-photo-realistic-lip-sync-from-text","slug":"obamanet-photo-realistic-lip-sync-from-text","title":"ObamaNet: Photo-realistic lip-sync from text","date":"2017-12-06","arxiv_id":"1801.01442","repositories_listed":1,"syntology":null},{"url":"/paper/massively-multilingual-neural-grapheme-to","slug":"massively-multilingual-neural-grapheme-to","title":"Massively Multilingual Neural Grapheme-to-Phoneme Conversion","date":"2017-08-04","arxiv_id":"1708.01464","repositories_listed":1,"syntology":null},{"url":"/paper/speech-coco-600k-visually-grounded-spoken","slug":"speech-coco-600k-visually-grounded-spoken","title":"SPEECH-COCO: 600k Visually Grounded Spoken Captions Aligned to MSCOCO Data Set","date":"2017-07-26","arxiv_id":"1707.08435","repositories_listed":1,"syntology":null},{"url":"/paper/dual-supervised-learning","slug":"dual-supervised-learning","title":"Dual Supervised Learning","date":"2017-07-03","arxiv_id":"1707.00415","repositories_listed":1,"syntology":null},{"url":"/paper/deep-voice-2-multi-speaker-neural-text-to","slug":"deep-voice-2-multi-speaker-neural-text-to","title":"Deep Voice 2: Multi-Speaker Neural Text-to-Speech","date":"2017-05-24","arxiv_id":"1705.08947","repositories_listed":1,"syntology":null},{"url":"/paper/rnn-approaches-to-text-normalization-a","slug":"rnn-approaches-to-text-normalization-a","title":"RNN Approaches to Text Normalization: A Challenge","date":"2016-10-31","arxiv_id":"1611.00068","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-distributions-with-linearizing","slug":"predicting-distributions-with-linearizing","title":"Predicting distributions with Linearizing Belief Networks","date":"2015-11-17","arxiv_id":"1511.05622","repositories_listed":1,"syntology":null},{"url":null,"slug":"hear-your-code-fail-voice-assisted-debugging","title":"Hear Your Code Fail, Voice-Assisted Debugging for Python","date":"2025-07-20","arxiv_id":"2507.15007","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonverbaltts-a-public-english-corpus-of-text","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","date":"2025-07-17","arxiv_id":"2507.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"p-808-multilingual-speech-enhancement-testing","title":"P.808 Multilingual Speech Enhancement Testing: Approach and Results of URGENT 2025 Challenge","date":"2025-07-15","arxiv_id":"2507.11306","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-evaluation-of-ai-powered-non","title":"An Empirical Evaluation of AI-Powered Non-Player Characters' Perceived Realism and Performance in Virtual Reality Environments","date":"2025-07-14","arxiv_id":"2507.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-leaderboards-for-large-scale","title":"Exploiting Leaderboards for Large-Scale Distribution of Malicious Models","date":"2025-07-11","arxiv_id":"2507.08983","repositories_listed":0,"syntology":null}],"record_sha256":"a5594bc9461fda9f20d7d1063d75a80c902db4187d68d8268deaaad985f17b2f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}