{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/4","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":13,"rows_per_page":100,"rows":[301,400],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/3","next":"/task/speech-synthesis/papers/5","papers":[{"url":"/paper/assem-vc-realistic-voice-conversion-by","slug":"assem-vc-realistic-voice-conversion-by","title":"Assem-VC: Realistic Voice Conversion by Assembling Modern Speech Synthesis Techniques","date":"2021-04-02","arxiv_id":"2104.00931","repositories_listed":1,"syntology":null},{"url":"/paper/styler-style-modeling-with-rapidity-and","slug":"styler-style-modeling-with-rapidity-and","title":"STYLER: Style Factor Modeling with Rapidity and Robustness via Speech Decomposition for Expressive and Controllable Neural Text to Speech","date":"2021-03-17","arxiv_id":"2103.09474","repositories_listed":1,"syntology":null},{"url":"/paper/handling-background-noise-in-neural-speech","slug":"handling-background-noise-in-neural-speech","title":"Handling Background Noise in Neural Speech Generation","date":"2021-02-23","arxiv_id":"2102.11906","repositories_listed":1,"syntology":null},{"url":"/paper/cdpam-contrastive-learning-for-perceptual","slug":"cdpam-contrastive-learning-for-perceptual","title":"CDPAM: Contrastive learning for perceptual audio similarity","date":"2021-02-09","arxiv_id":"2102.05109","repositories_listed":1,"syntology":null},{"url":"/paper/text-free-image-to-speech-synthesis-using","slug":"text-free-image-to-speech-synthesis-using","title":"Text-Free Image-to-Speech Synthesis Using Learned Segmental Units","date":"2020-12-31","arxiv_id":"2012.15454","repositories_listed":1,"syntology":null},{"url":"/paper/multi-view-temporal-alignment-for-non","slug":"multi-view-temporal-alignment-for-non","title":"Multi-view Temporal Alignment for Non-parallel Articulatory-to-Acoustic Speech Synthesis","date":"2020-12-30","arxiv_id":"2012.15184","repositories_listed":1,"syntology":null},{"url":"/paper/deeptalk-vocal-style-encoding-for-speaker","slug":"deeptalk-vocal-style-encoding-for-speaker","title":"DeepTalk: Vocal Style Encoding for Speaker Recognition and Speech Synthesis","date":"2020-12-09","arxiv_id":"2012.05084","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-disentanglement-of-speaker","slug":"adversarial-disentanglement-of-speaker","title":"Adversarial Disentanglement of Speaker Representation for Attribute-Driven Privacy Preservation","date":"2020-12-08","arxiv_id":"2012.04454","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-url-segmentation-with-1","slug":"semi-supervised-url-segmentation-with-1","title":"Semi-supervised URL Segmentation with Recurrent Neural Networks Pre-trained on Knowledge Graph Entities","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tfgan-time-and-frequency-domain-based","slug":"tfgan-time-and-frequency-domain-based","title":"TFGAN: Time and Frequency Domain Based Generative Adversarial Network for High-fidelity Speech Synthesis","date":"2020-11-24","arxiv_id":"2011.12206","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-prosody-modeling-for-non","slug":"hierarchical-prosody-modeling-for-non","title":"Hierarchical Prosody Modeling for Non-Autoregressive Speech Synthesis","date":"2020-11-12","arxiv_id":"2011.06465","repositories_listed":1,"syntology":null},{"url":"/paper/wave-tacotron-spectrogram-free-end-to-end","slug":"wave-tacotron-spectrogram-free-end-to-end","title":"Wave-Tacotron: Spectrogram-free end-to-end text-to-speech synthesis","date":"2020-11-06","arxiv_id":"2011.03568","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-url-segmentation-with","slug":"semi-supervised-url-segmentation-with","title":"Semi-supervised URL Segmentation with Recurrent Neural NetworksPre-trained on Knowledge Graph Entities","date":"2020-11-05","arxiv_id":"2011.03138","repositories_listed":1,"syntology":null},{"url":"/paper/effective-deep-learning-models-for-automatic","slug":"effective-deep-learning-models-for-automatic","title":"Effective Deep Learning Models for Automatic Diacritization of Arabic Text","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-disentangled-phone-and-speaker","slug":"learning-disentangled-phone-and-speaker","title":"Learning Disentangled Phone and Speaker Representations in a Semi-Supervised VQ-VAE Paradigm","date":"2020-10-21","arxiv_id":"2010.10727","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-disentangled-phone-and-speaker#ran","syntology_url":"https://syntology.ai/paper/2010.10727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10727"}},"official":{"repos":["rhoposit/icassp2021"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-natural-bilingual-and-code-switched","slug":"towards-natural-bilingual-and-code-switched","title":"Towards Natural Bilingual and Code-Switched Speech Synthesis Based on Mix of Monolingual Recordings and Cross-Lingual Voice Conversion","date":"2020-10-16","arxiv_id":"2010.08136","repositories_listed":1,"syntology":null},{"url":"/paper/digital-voicing-of-silent-speech","slug":"digital-voicing-of-silent-speech","title":"Digital Voicing of Silent Speech","date":"2020-10-06","arxiv_id":"2010.02960","repositories_listed":1,"syntology":null},{"url":"/paper/jsss-free-japanese-speech-corpus-for","slug":"jsss-free-japanese-speech-corpus-for","title":"JSSS: free Japanese speech corpus for summarization and simplification","date":"2020-10-05","arxiv_id":"2010.01793","repositories_listed":1,"syntology":null},{"url":"/paper/dynamical-variational-autoencoders-a","slug":"dynamical-variational-autoencoders-a","title":"Dynamical Variational Autoencoders: A Comprehensive Review","date":"2020-08-28","arxiv_id":"2008.12595","repositories_listed":1,"syntology":null},{"url":"/paper/laughter-synthesis-combining-seq2seq-modeling","slug":"laughter-synthesis-combining-seq2seq-modeling","title":"Laughter Synthesis: Combining Seq2seq modeling with Transfer Learning","date":"2020-08-20","arxiv_id":"2008.09483","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-speech-intelligibility-in-text-to","slug":"enhancing-speech-intelligibility-in-text-to","title":"Enhancing Speech Intelligibility in Text-To-Speech Synthesis using Speaking Style Conversion","date":"2020-08-13","arxiv_id":"2008.05809","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-speech-intelligibility-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2008.05809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05809"}},"official":null}},{"url":"/paper/attentron-few-shot-text-to-speech-utilizing-1","slug":"attentron-few-shot-text-to-speech-utilizing-1","title":"Attentron: Few-Shot Text-to-Speech Utilizing Attention-Based Variable-Length Embedding","date":"2020-08-12","arxiv_id":"2005.08484","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/attentron-few-shot-text-to-speech-utilizing-1#ran","syntology_url":"https://syntology.ai/paper/2005.08484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08484"}},"official":null}},{"url":"/paper/improving-opus-low-bit-rate-quality-with","slug":"improving-opus-low-bit-rate-quality-with","title":"Improving Opus Low Bit Rate Quality with Neural Speech Synthesis","date":"2020-08-10","arxiv_id":"1905.04628","repositories_listed":1,"syntology":null},{"url":"/paper/speaker-conditional-wavernn-towards-universal","slug":"speaker-conditional-wavernn-towards-universal","title":"Speaker Conditional WaveRNN: Towards Universal Neural Vocoder for Unseen Speaker and Recording Conditions","date":"2020-08-09","arxiv_id":"2008.05289","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speaker-conditional-wavernn-towards-universal#ran","syntology_url":"https://syntology.ai/paper/2008.05289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.05289"}},"official":null}},{"url":"/paper/phonological-features-for-0-shot-multilingual","slug":"phonological-features-for-0-shot-multilingual","title":"Phonological Features for 0-shot Multilingual Speech Synthesis","date":"2020-08-06","arxiv_id":"2008.04107","repositories_listed":1,"syntology":null},{"url":"/paper/one-model-many-languages-meta-learning-for","slug":"one-model-many-languages-meta-learning-for","title":"One Model, Many Languages: Meta-learning for Multilingual Text-to-Speech","date":"2020-08-03","arxiv_id":"2008.00768","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-quantum-neural-networks","slug":"recurrent-quantum-neural-networks","title":"Recurrent Quantum Neural Networks","date":"2020-06-25","arxiv_id":"2006.14619","repositories_listed":1,"syntology":null},{"url":"/paper/nanoflow-scalable-normalizing-flows-with","slug":"nanoflow-scalable-normalizing-flows-with","title":"NanoFlow: Scalable Normalizing Flows with Sublinear Parameter Complexity","date":"2020-06-11","arxiv_id":"2006.06280","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/nanoflow-scalable-normalizing-flows-with#ran","syntology_url":"https://syntology.ai/paper/2006.06280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06280"}},"official":{"repos":["L0SG/NanoFlow"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/wavenode-a-continuous-normalizing-flow-for","slug":"wavenode-a-continuous-normalizing-flow-for","title":"WaveNODE: A Continuous Normalizing Flow for Speech Synthesis","date":"2020-06-08","arxiv_id":"2006.04598","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/wavenode-a-continuous-normalizing-flow-for#ran","syntology_url":"https://syntology.ai/paper/2006.04598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04598"}},"official":null}},{"url":"/paper/polydl-polyhedral-optimizations-for-creation","slug":"polydl-polyhedral-optimizations-for-creation","title":"PolyDL: Polyhedral Optimizations for Creation of High Performance DL primitives","date":"2020-06-02","arxiv_id":"2006.02230","repositories_listed":1,"syntology":null},{"url":"/paper/learning-individual-speaking-styles-for","slug":"learning-individual-speaking-styles-for","title":"Learning Individual Speaking Styles for Accurate Lip to Speech Synthesis","date":"2020-05-17","arxiv_id":"2005.08209","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-individual-speaking-styles-for#ran","syntology_url":"https://syntology.ai/paper/2005.08209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08209"}},"official":{"repos":["Rudrabha/Lip2Wav"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-speech-synthesis-applied-to","slug":"end-to-end-speech-synthesis-applied-to","title":"TTS-Portuguese Corpus: a corpus for speech synthesis in Brazilian Portuguese","date":"2020-05-11","arxiv_id":"2005.05144","repositories_listed":1,"syntology":null},{"url":"/paper/from-speaker-verification-to-multispeaker","slug":"from-speaker-verification-to-multispeaker","title":"From Speaker Verification to Multispeaker Speech Synthesis, Deep Transfer with Feedback Constraint","date":"2020-05-10","arxiv_id":"2005.04587","repositories_listed":1,"syntology":null},{"url":"/paper/can-speaker-augmentation-improve-multi","slug":"can-speaker-augmentation-improve-multi","title":"Can Speaker Augmentation Improve Multi-Speaker End-to-End TTS?","date":"2020-05-04","arxiv_id":"2005.01245","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-speaker-augmentation-improve-multi#ran","syntology_url":"https://syntology.ai/paper/2005.01245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.01245"}},"official":{"repos":["nii-yamagishilab/multi-speaker-tacotron"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-technology-programme-for-icelandic","slug":"language-technology-programme-for-icelandic","title":"Language Technology Programme for Icelandic 2019-2023","date":"2020-03-20","arxiv_id":"2003.09244","repositories_listed":1,"syntology":null},{"url":"/paper/perception-of-prosodic-variation-for-speech","slug":"perception-of-prosodic-variation-for-speech","title":"Perception of prosodic variation for speech synthesis using an unsupervised discrete representation of F0","date":"2020-03-14","arxiv_id":"2003.06686","repositories_listed":1,"syntology":null},{"url":"/paper/a-neuro-ai-interface-for-evaluating","slug":"a-neuro-ai-interface-for-evaluating","title":"A Neuro-AI Interface for Evaluating Generative Adversarial Networks","date":"2020-03-05","arxiv_id":"2003.03193","repositories_listed":1,"syntology":null},{"url":"/paper/improving-lpcnet-based-text-to-speech-with","slug":"improving-lpcnet-based-text-to-speech-with","title":"Improving LPCNet-based Text-to-Speech with Linear Prediction-structured Mixture Density Network","date":"2020-01-31","arxiv_id":"2001.11686","repositories_listed":1,"syntology":null},{"url":"/paper/a-resource-for-computational-experiments-on","slug":"a-resource-for-computational-experiments-on","title":"A Resource for Computational Experiments on Mapudungun","date":"2019-12-04","arxiv_id":"1912.01772","repositories_listed":1,"syntology":null},{"url":"/paper/jejueo-datasets-for-machine-translation-and","slug":"jejueo-datasets-for-machine-translation-and","title":"Jejueo Datasets for Machine Translation and Speech Synthesis","date":"2019-11-27","arxiv_id":"1911.12071","repositories_listed":1,"syntology":null},{"url":"/paper/independent-and-automatic-evaluation-of","slug":"independent-and-automatic-evaluation-of","title":"Independent and automatic evaluation of acoustic-to-articulatory inversion models","date":"2019-11-15","arxiv_id":"1911.06573","repositories_listed":1,"syntology":null},{"url":"/paper/what-does-a-network-layer-hear-analyzing","slug":"what-does-a-network-layer-hear-analyzing","title":"What does a network layer hear? Analyzing hidden representations of end-to-end ASR through speech synthesis","date":"2019-11-04","arxiv_id":"1911.01102","repositories_listed":1,"syntology":null},{"url":"/paper/spoofing-speaker-verification-systems-with","slug":"spoofing-speaker-verification-systems-with","title":"Spoofing Speaker Verification Systems with Deep Multi-speaker Text-to-speech Synthesis","date":"2019-10-29","arxiv_id":"1910.13054","repositories_listed":1,"syntology":null},{"url":"/paper/disentangling-speech-and-non-speech","slug":"disentangling-speech-and-non-speech","title":"Disentangling Speech and Non-Speech Components for Building Robust Acoustic Models from Found Data","date":"2019-09-25","arxiv_id":"1909.11727","repositories_listed":1,"syntology":null},{"url":"/paper/unpaired-image-to-speech-synthesis-with","slug":"unpaired-image-to-speech-synthesis-with","title":"Unpaired Image-to-Speech Synthesis with Multimodal Information Bottleneck","date":"2019-08-19","arxiv_id":"1908.07094","repositories_listed":1,"syntology":null},{"url":"/paper/deep-residual-neural-networks-for-audio","slug":"deep-residual-neural-networks-for-audio","title":"Deep Residual Neural Networks for Audio Spoofing Detection","date":"2019-06-30","arxiv_id":"1907.00501","repositories_listed":1,"syntology":null},{"url":"/paper/using-generative-modelling-to-produce-varied","slug":"using-generative-modelling-to-produce-varied","title":"Using generative modelling to produce varied intonation for speech synthesis","date":"2019-06-10","arxiv_id":"1906.04233","repositories_listed":1,"syntology":null},{"url":"/paper/effective-use-of-variational-embedding","slug":"effective-use-of-variational-embedding","title":"Effective Use of Variational Embedding Capacity in Expressive End-to-End Speech Synthesis","date":"2019-06-08","arxiv_id":"1906.03402","repositories_listed":1,"syntology":null},{"url":"/paper/effective-parameter-estimation-methods-for-an","slug":"effective-parameter-estimation-methods-for-an","title":"Effective parameter estimation methods for an ExcitNet model in generative text-to-speech systems","date":"2019-05-21","arxiv_id":"1905.08486","repositories_listed":1,"syntology":null},{"url":"/paper/neuroscore-a-brain-inspired-evaluation-metric","slug":"neuroscore-a-brain-inspired-evaluation-metric","title":"Synthetic-Neuroscore: Using A Neuro-AI Interface for Evaluating Generative Adversarial Networks","date":"2019-05-10","arxiv_id":"1905.04243","repositories_listed":1,"syntology":null},{"url":"/paper/spoof-detection-using-x-vector-and-feature","slug":"spoof-detection-using-x-vector-and-feature","title":"Spoof detection using time-delay shallow neural network and feature switching","date":"2019-04-16","arxiv_id":"1904.07453","repositories_listed":1,"syntology":null},{"url":"/paper/direct-speech-to-speech-translation-with-a","slug":"direct-speech-to-speech-translation-with-a","title":"Direct speech-to-speech translation with a sequence-to-sequence model","date":"2019-04-12","arxiv_id":"1904.06037","repositories_listed":1,"syntology":null},{"url":"/paper/stc-antispoofing-systems-for-the-asvspoof2019","slug":"stc-antispoofing-systems-for-the-asvspoof2019","title":"STC Antispoofing Systems for the ASVspoof2019 Challenge","date":"2019-04-11","arxiv_id":"1904.05576","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stc-antispoofing-systems-for-the-asvspoof2019#ran","syntology_url":"https://syntology.ai/paper/1904.05576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.05576"}},"official":null}},{"url":"/paper/gelp-gan-excited-liner-prediction-for-speech","slug":"gelp-gan-excited-liner-prediction-for-speech","title":"GELP: GAN-Excited Linear Prediction for Speech Synthesis from Mel-spectrogram","date":"2019-04-08","arxiv_id":"1904.03976","repositories_listed":1,"syntology":null},{"url":"/paper/in-other-news-a-bi-style-text-to-speech-model","slug":"in-other-news-a-bi-style-text-to-speech-model","title":"In Other News: A Bi-style Text-to-speech Model for Synthesizing Newscaster Voice with Limited Data","date":"2019-04-04","arxiv_id":"1904.02790","repositories_listed":1,"syntology":null},{"url":"/paper/visualization-and-interpretation-of-latent","slug":"visualization-and-interpretation-of-latent","title":"Visualization and Interpretation of Latent Spaces for Controlling Expressive Speech Synthesis through Audio Analysis","date":"2019-03-27","arxiv_id":"1903.11570","repositories_listed":1,"syntology":null},{"url":"/paper/robust-and-fine-grained-prosody-control-of","slug":"robust-and-fine-grained-prosody-control-of","title":"Robust and fine-grained prosody control of end-to-end speech synthesis","date":"2018-11-06","arxiv_id":"1811.02122","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-and-fine-grained-prosody-control-of#ran","syntology_url":"https://syntology.ai/paper/1811.02122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.02122"}},"official":null}},{"url":"/paper/investigation-of-enhanced-tacotron-text-to","slug":"investigation-of-enhanced-tacotron-text-to","title":"Investigation of enhanced Tacotron text-to-speech synthesis systems with self-attention for pitch accent language","date":"2018-10-29","arxiv_id":"1810.11960","repositories_listed":1,"syntology":null},{"url":"/paper/learning-pronunciation-from-a-foreign-1","slug":"learning-pronunciation-from-a-foreign-1","title":"Learning pronunciation from a foreign language in speech synthesis networks","date":"2018-10-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/the-emotional-voices-database-towards","slug":"the-emotional-voices-database-towards","title":"The Emotional Voices Database: Towards Controlling the Emotion Dimension in Voice Generation Systems","date":"2018-06-25","arxiv_id":"1806.09514","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/the-emotional-voices-database-towards#ran","syntology_url":"https://syntology.ai/paper/1806.09514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.09514"}},"official":{"repos":["numediart/EmoV-DB"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-variational-prosody-model-for-the","slug":"a-variational-prosody-model-for-the","title":"A Variational Prosody Model for the decomposition and synthesis of speech prosody","date":"2018-06-22","arxiv_id":"1806.08685","repositories_listed":1,"syntology":null},{"url":"/paper/fonbund-a-library-for-combining-cross-lingual","slug":"fonbund-a-library-for-combining-cross-lingual","title":"FonBund: A Library for Combining Cross-lingual Phonological Segment Data","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/speech-waveform-synthesis-from-mfcc-sequences","slug":"speech-waveform-synthesis-from-mfcc-sequences","title":"Speech waveform synthesis from MFCC sequences with generative adversarial networks","date":"2018-04-03","arxiv_id":"1804.00920","repositories_listed":1,"syntology":null},{"url":"/paper/jsut-corpus-free-large-scale-japanese-speech","slug":"jsut-corpus-free-large-scale-japanese-speech","title":"JSUT corpus: free large-scale Japanese speech corpus for end-to-end speech synthesis","date":"2017-10-28","arxiv_id":"1711.00354","repositories_listed":1,"syntology":null},{"url":"/paper/using-hyperlinks-to-improve-multilingual","slug":"using-hyperlinks-to-improve-multilingual","title":"Using hyperlinks to improve multilingual partial parsers","date":"2017-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-voice-2-multi-speaker-neural-text-to","slug":"deep-voice-2-multi-speaker-neural-text-to","title":"Deep Voice 2: Multi-Speaker Neural Text-to-Speech","date":"2017-05-24","arxiv_id":"1705.08947","repositories_listed":1,"syntology":null},{"url":null,"slug":"nonverbaltts-a-public-english-corpus-of-text","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","date":"2025-07-17","arxiv_id":"2507.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-quality-assessment-model-based-on","title":"Speech Quality Assessment Model Based on Mixture of Experts: System-Level Performance Enhancement and Utterance-Level Challenge Analysis","date":"2025-07-08","arxiv_id":"2507.06116","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-machine-learning-framework-for","title":"A Hybrid Machine Learning Framework for Optimizing Crop Selection via Agronomic and Economic Forecasting","date":"2025-07-06","arxiv_id":"2507.08832","repositories_listed":0,"syntology":null},{"url":null,"slug":"opuslm-a-family-of-open-unified-speech","title":"OpusLM: A Family of Open Unified Speech Language Models","date":"2025-06-21","arxiv_id":"2506.17611","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-accurate-and-revised-version-of-optical","title":"An accurate and revised version of optical character recognition-based speech synthesis using LabVIEW","date":"2025-06-18","arxiv_id":"2506.15029","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-flat-to-feeling-a-feasibility-and-impact","title":"From Flat to Feeling: A Feasibility and Impact Study on Dynamic Facial Emotions in AI-Generated Avatars","date":"2025-06-16","arxiv_id":"2506.13477","repositories_listed":0,"syntology":null},{"url":null,"slug":"s2st-omni-an-efficient-and-scalable","title":"S2ST-Omni: An Efficient and Scalable Multilingual Speech-to-Speech Translation Framework via Seamless Speech-Text Alignment and Streaming Speech Generation","date":"2025-06-11","arxiv_id":"2506.11160","repositories_listed":0,"syntology":null},{"url":null,"slug":"umbratts-adapting-text-to-speech-to","title":"UmbraTTS: Adapting Text-to-Speech to Environmental Contexts with Flow Matching","date":"2025-06-11","arxiv_id":"2506.09874","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-voices-generating-a-roll-video-from","title":"Seeing Voices: Generating A-Roll Video from Audio with Mirage","date":"2025-06-09","arxiv_id":"2506.08279","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-data-augmentation-approach-for","title":"A Novel Data Augmentation Approach for Automatic Speaking Assessment on Opinion Expressions","date":"2025-06-04","arxiv_id":"2506.04077","repositories_listed":0,"syntology":null},{"url":null,"slug":"hifitts-2-a-large-scale-high-bandwidth-speech","title":"HiFiTTS-2: A Large-Scale High Bandwidth Speech Dataset","date":"2025-06-04","arxiv_id":"2506.04152","repositories_listed":0,"syntology":null},{"url":null,"slug":"capspeech-enabling-downstream-applications-in","title":"CapSpeech: Enabling Downstream Applications in Style-Captioned Text-to-Speech","date":"2025-06-03","arxiv_id":"2506.02863","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-unseen-emotion-zero-shot-expressive","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","date":"2025-06-03","arxiv_id":"2506.02742","repositories_listed":0,"syntology":null},{"url":null,"slug":"salf-mos-speaker-agnostic-latent-features","title":"SALF-MOS: Speaker Agnostic Latent Features Downsampled for MOS Prediction","date":"2025-06-02","arxiv_id":"2506.02082","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-activation-editing-for-post","title":"Counterfactual Activation Editing for Post-hoc Prosody and Mispronunciation Correction in TTS Models","date":"2025-06-01","arxiv_id":"2506.00832","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-thought-training-for-open-e2e-spoken","title":"Chain-of-Thought Training for Open E2E Spoken Dialogue Systems","date":"2025-05-31","arxiv_id":"2506.00722","repositories_listed":0,"syntology":null},{"url":null,"slug":"binauralflow-a-causal-and-streamable-approach","title":"BinauralFlow: A Causal and Streamable Approach for High-Quality Binaural Speech Synthesis with Flow Matching Models","date":"2025-05-28","arxiv_id":"2505.22865","repositories_listed":0,"syntology":null},{"url":"/paper/arvoice-a-multi-speaker-dataset-for-arabic","slug":"arvoice-a-multi-speaker-dataset-for-arabic","title":"ArVoice: A Multi-Speaker Dataset for Arabic Speech Synthesis","date":"2025-05-26","arxiv_id":"2505.20506","repositories_listed":0,"syntology":null},{"url":null,"slug":"diemo-tts-disentangled-emotion","title":"DiEmo-TTS: Disentangled Emotion Representations via Self-Supervised Distillation for Cross-Speaker Emotion Transfer in Text-to-Speech","date":"2025-05-26","arxiv_id":"2505.19687","repositories_listed":0,"syntology":null},{"url":null,"slug":"gsa-tts-toward-zero-shot-speech-synthesis","title":"GSA-TTS : Toward Zero-Shot Speech Synthesis based on Gradual Style Adaptor","date":"2025-05-26","arxiv_id":"2505.19384","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-streaming-text-to-speech-synthesis","title":"Zero-Shot Streaming Text to Speech Synthesis with Transducer and Auto-Regressive Modeling","date":"2025-05-26","arxiv_id":"2505.19669","repositories_listed":0,"syntology":null},{"url":null,"slug":"revival-with-voice-multi-modal-controllable","title":"Revival with Voice: Multi-modal Controllable Text-to-Speech Synthesis","date":"2025-05-25","arxiv_id":"2505.18972","repositories_listed":0,"syntology":null},{"url":null,"slug":"rasmalai-resources-for-adaptive-speech","title":"RASMALAI: Resources for Adaptive Speech Modeling in Indian Languages with Accents and Intonations","date":"2025-05-24","arxiv_id":"2505.18609","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-autoregressive-speech-synthesis","title":"Accelerating Autoregressive Speech Synthesis Inference With Speech Speculative Decoding","date":"2025-05-21","arxiv_id":"2505.15380","repositories_listed":0,"syntology":null},{"url":null,"slug":"miku-pal-an-automated-and-standardized-multi","title":"MIKU-PAL: An Automated and Standardized Multi-Modal Method for Speech Paralinguistic and Affect Labeling","date":"2025-05-21","arxiv_id":"2505.15772","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmsd-tts-few-shot-multi-speaker-multi-dialect","title":"FMSD-TTS: Few-shot Multi-Speaker Multi-Dialect Text-to-Speech Synthesis for Ü-Tsang, Amdo and Kham Speech Dataset Generation","date":"2025-05-20","arxiv_id":"2505.14351","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairwise-evaluation-of-accent-similarity-in","title":"Pairwise Evaluation of Accent Similarity in Speech Synthesis","date":"2025-05-20","arxiv_id":"2505.14410","repositories_listed":0,"syntology":null},{"url":null,"slug":"ozspeech-one-step-zero-shot-speech-synthesis","title":"OZSpeech: One-step Zero-shot Speech Synthesis with Learned-Prior-Conditioned Flow Matching","date":"2025-05-19","arxiv_id":"2505.12800","repositories_listed":0,"syntology":null},{"url":null,"slug":"rovo-robust-voice-protection-against","title":"RoVo: Robust Voice Protection Against Unauthorized Speech Synthesis with Embedding-Level Perturbations","date":"2025-05-19","arxiv_id":"2505.12686","repositories_listed":0,"syntology":null},{"url":null,"slug":"shallow-flow-matching-for-coarse-to-fine-text","title":"Shallow Flow Matching for Coarse-to-Fine Text-to-Speech Synthesis","date":"2025-05-18","arxiv_id":"2505.12226","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10599","title":"UDDETTS: Unifying Discrete and Dimensional Emotions for Controllable Emotional Text-to-Speech","date":"2025-05-15","arxiv_id":"2505.10599","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpn-gan-inducing-periodic-activations-in","title":"DPN-GAN: Inducing Periodic Activations in Generative Adversarial Networks for High-Fidelity Audio Synthesis","date":"2025-05-14","arxiv_id":"2505.09091","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-self-supervised-features-for","title":"Investigating self-supervised features for expressive, multilingual voice conversion","date":"2025-05-13","arxiv_id":"2505.08278","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-end-to-end-text-to-speech","title":"Lightweight End-to-end Text-to-speech Synthesis for low resource on-device applications","date":"2025-05-12","arxiv_id":"2505.07701","repositories_listed":0,"syntology":null}],"record_sha256":"cd03ea907cdfb4ae6674d6c464a8db3820d2c947acbc98c3fc180dbf87902d68","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}