{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/high-fidelity-speech-synthesis-with-1","title":"High Fidelity Speech Synthesis with Adversarial Networks","arxiv_id":"1909.11646","date":"2019-09-25","proceeding":"ICLR 2020 1","authors":["Mikołaj Bińkowski","Jeff Donahue","Sander Dieleman","Aidan Clark","Erich Elsen","Norman Casagrande","Luis C. Cobo","Karen Simonyan"],"abstract":"Generative adversarial networks have seen rapid development in recent years and have led to remarkable improvements in generative modelling of images. However, their application in the audio domain has received limited attention, and autoregressive models, such as WaveNet, remain the state of the art in generative modelling of audio signals such as human speech. To address this paucity, we introduce GAN-TTS, a Generative Adversarial Network for Text-to-Speech. Our architecture is composed of a conditional feed-forward generator producing raw speech audio, and an ensemble of discriminators which operate on random windows of different sizes. The discriminators analyse the audio both in terms of general realism, as well as how well the audio corresponds to the utterance that should be pronounced. To measure the performance of GAN-TTS, we employ both subjective human evaluation (MOS - Mean Opinion Score), as well as novel quantitative metrics (Fr\\'echet DeepSpeech Distance and Kernel DeepSpeech Distance), which we find to be well correlated with MOS. We show that GAN-TTS is capable of generating high-fidelity speech with naturalness comparable to the state-of-the-art models, and unlike autoregressive models, it is highly parallelisable thanks to an efficient feed-forward generator. Listen to GAN-TTS reading this abstract at https://storage.googleapis.com/deepmind-media/research/abstract.wav.","url_abs":"https://arxiv.org/abs/1909.11646v2","url_pdf":"https://arxiv.org/pdf/1909.11646v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"high-fidelity-speech-synthesis-with-1","repo_url":"https://github.com/mbinkowski/DeepSpeechDistances","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"high-fidelity-speech-synthesis-with-1","repo_url":"https://github.com/izzajalandoni/tts_models","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"BSD-3-Clause"}},{"paper_slug":"high-fidelity-speech-synthesis-with-1","repo_url":"https://github.com/yanggeng1995/GAN-TTS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":null,"task_name":"Generative Adversarial Network"},{"task_slug":"speech-synthesis","task_name":"Speech Synthesis"},{"task_slug":"text-to-speech","task_name":"Text to Speech"},{"task_slug":"high","task_name":"Vocal Bursts Intensity Prediction"},{"task_slug":"text-to-speech-1","task_name":"text-to-speech"}],"methods":[{"method_slug":"1x1-convolution","method_name":"1x1 Convolution"},{"method_slug":"adam","method_name":"Adam"},{"method_slug":"average-pooling","method_name":"Average Pooling"},{"method_slug":"batch-normalization","method_name":"Batch Normalization"},{"method_slug":"conditional-batch-normalization","method_name":"Conditional Batch Normalization"},{"method_slug":"conditional-dblock","method_name":"Conditional DBlock"},{"method_slug":"convolution","method_name":"Convolution"},{"method_slug":"dblock","method_name":"DBlock"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dilated-causal-convolution","method_name":"Dilated Causal Convolution"},{"method_slug":"dilated-convolution","method_name":"Dilated Convolution"},{"method_slug":"feedforward-network","method_name":"Feedforward Network"},{"method_slug":"gan-tts","method_name":"GAN-TTS"},{"method_slug":"gblock","method_name":"GBlock"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"mixture-of-logistic-distributions","method_name":"Mixture of Logistic Distributions"},{"method_slug":"multiple-random-window-discriminator","method_name":"Multiple Random Window Discriminator"},{"method_slug":"off-diagonal-orthogonal-regularization","method_name":"Off-Diagonal Orthogonal Regularization"},{"method_slug":"orthogonal-regularization","method_name":"Orthogonal Regularization"},{"method_slug":"relu","method_name":"ReLU"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"spectral-normalization","method_name":"Spectral Normalization"},{"method_slug":"tanh-activation","method_name":"Tanh Activation"},{"method_slug":"wavenet","method_name":"WaveNet"}],"datasets_introduced":[],"methods_introduced":[{"slug":"conditional-dblock","name":"Conditional DBlock","full_name":"Conditional DBlock"},{"slug":"dblock","name":"DBlock","full_name":"DBlock"},{"slug":"gan-tts","name":"GAN-TTS","full_name":"GAN-TTS"},{"slug":"gblock","name":"GBlock","full_name":"GBlock"},{"slug":"multiple-random-window-discriminator","name":"Multiple Random Window Discriminator","full_name":"Multiple Random Window Discriminator"}],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1909.11646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11646"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mbinkowski/DeepSpeechDistances","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/yanggeng1995/GAN-TTS","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/izzajalandoni/tts_models","reach":{"status":"ok","spdx":"BSD-3-Clause"}}],"summary":{"ran_draft_wrong":2,"unverified":3},"by_repo_kind":{"official":{"samples":1,"ran":0,"repositories":1},"listed":{"samples":3,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"83e2c862f7374ac7","entry":"dynamic_range_compression","repo":"izzajalandoni/tts_models","repo_kind":"listed","path":"audio_processing.py","file_url":"https://github.com/izzajalandoni/tts_models/blob/HEAD/audio_processing.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"83e2c862f7374ac7"}},{"code_sha256_prefix":"9f9ec26d4cdfdf7d","entry":"griffin_lim","repo":"izzajalandoni/tts_models","repo_kind":"listed","path":"audio_processing.py","file_url":"https://github.com/izzajalandoni/tts_models/blob/HEAD/audio_processing.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"9f9ec26d4cdfdf7d"}},{"code_sha256_prefix":"f7bf9f0265e33cf0","entry":"apply_gradient_allreduce","repo":"izzajalandoni/tts_models","repo_kind":"listed","path":"distributed.py","file_url":"https://github.com/izzajalandoni/tts_models/blob/HEAD/distributed.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"f7bf9f0265e33cf0"}},{"code_sha256_prefix":"f443ef0155d2d898","entry":"load_checkpoint","repo":null,"repo_kind":null,"path":null,"file_url":null,"link_basis":"identical_code_first_harvested_elsewhere","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":null,"inline_ok":false,"mcp_get_code":{"code_sha256":"f443ef0155d2d898"}},{"code_sha256_prefix":"d427fbdc46645d33","entry":"normalize_signal","repo":"mbinkowski/DeepSpeechDistances","repo_kind":"official","path":"preprocessing.py","file_url":"https://github.com/mbinkowski/DeepSpeechDistances/blob/HEAD/preprocessing.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d427fbdc46645d33"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}