{"url":"/method/wavegrad","slug":"wavegrad","name":"WaveGrad","full_name":"WaveGrad","full_name_withheld":false,"description_markdown":"**WaveGrad** is a conditional model for waveform generation through estimating gradients of the data density. This model is built on the prior work on score matching and diffusion probabilistic models. It starts from Gaussian white noise and iteratively refines the signal via a gradient-based sampler conditioned on the mel-spectrogram. WaveGrad is non-autoregressive, and requires only a constant number of generation steps during inference. It can use as few as 6 iterations to generate high fidelity audio samples.","description_state":"present","introduced_year":null,"introduced_by":{"title":"WaveGrad: Estimating Gradients for Waveform Generation","paper":"/paper/wavegrad-estimating-gradients-for-waveform","first_author":"Nanxin Chen","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/wavegrad-estimating-gradients-for-waveform"},"source":{"url":"https://arxiv.org/abs/2009.00713v2","title":"WaveGrad: Estimating Gradients for Waveform Generation","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/lmnt-com/wavegrad/blob/master/src/wavegrad/model.py","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Audio","area_id":"audio","collection":"Generative Audio Models","url":"/methods/category/generative-audio-models","pwc_aliases":[]}],"n_papers_tagged":7,"archive_num_papers":7,"papers_newest_first":[{"paper":null,"title":"GLA-Grad: A Griffin-Lim Extended Waveform Generation Diffusion Model","date":"2024-02-09","arxiv_id":"2402.15516","n_code_links":0,"syntology":null},{"paper":"/paper/bddm-bilateral-denoising-diffusion-models-for-1","title":"BDDM: Bilateral Denoising Diffusion Models for Fast and High-Quality Speech Synthesis","date":"2022-03-25","arxiv_id":"2203.13508","n_code_links":1,"syntology":{"ran":5,"of":6,"unverified":1,"pointer_only":0}},{"paper":null,"title":"InferGrad: Improving Diffusion Models for Vocoder by Considering Inference in Training","date":"2022-02-08","arxiv_id":"2202.03751","n_code_links":0,"syntology":null},{"paper":null,"title":"Quasi-Taylor Samplers for Diffusion Generative Models based on Ideal Derivatives","date":"2021-12-26","arxiv_id":"2112.13339","n_code_links":0,"syntology":null},{"paper":"/paper/vocbench-a-neural-vocoder-benchmark-for","title":"VocBench: A Neural Vocoder Benchmark for Speech Synthesis","date":"2021-12-06","arxiv_id":"2112.03099","n_code_links":1,"syntology":null},{"paper":"/paper/wavegrad-2-iterative-refinement-for-text-to","title":"WaveGrad 2: Iterative Refinement for Text-to-Speech Synthesis","date":"2021-06-17","arxiv_id":"2106.09660","n_code_links":3,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":0}},{"paper":"/paper/wavegrad-estimating-gradients-for-waveform","title":"WaveGrad: Estimating Gradients for Waveform Generation","date":"2020-09-02","arxiv_id":"2009.00713","n_code_links":7,"syntology":{"ran":1,"of":3,"unverified":2,"pointer_only":1}}],"papers_shown":7,"tasks":[{"task":"/task/speech-synthesis","name":"Speech Synthesis","papers":5},{"task":"/task/denoising","name":"Denoising","papers":2},{"task":"/task/image-generation","name":"Image Generation","papers":2},{"task":"/task/text-to-speech-synthesis","name":"Text-To-Speech Synthesis","papers":2},{"task":"/task/text-to-speech","name":"Text to Speech","papers":1},{"task":"/task/text-to-speech-1","name":"text-to-speech","papers":1}],"tasks_shown":6,"n_tasks":6,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":3},{"year":"2022","papers":2},{"year":"2024","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/wavegrad"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}