{"url":"/dataset/demand","name":"VoiceBank + DEMAND","full_name":"Noisy speech database for training speech enhancement algorithms and TTS models","description_markdown":"VoiceBank+DEMAND is a noisy speech database for training speech enhancement algorithms and TTS models. The database was designed to train and test speech enhancement methods that operate at 48kHz. A more detailed description can be found in the paper associated with the database. Some of the noises were obtained from the Demand database, available here: http://parole.loria.fr/DEMAND/ . The speech database was obtained from the Voice Banking Corpus, available here: http://homepages.inf.ed.ac.uk/jyamagis/release/VCTK-Corpus.tar.gz .","description_withheld":null,"homepage":"https://datashare.ed.ac.uk/handle/10283/2791","introduced_date":"2016-09-12","introduced_date_note":null,"introduced_by":{"paper":null,"title":"The Diverse Environments Multi-channel Acoustic Noise Database (DEMAND): A database of multichannel environmental noise recordings","first_author":null,"url":"https://asa.scitation.org/doi/abs/10.1121/1.4799597"},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Speech Enhancement","url":"/task/speech-enhancement","datasets_with_task":"/datasets/task/speech-enhancement"}],"languages":[],"variants":["VoiceBank + DEMAND"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/JacobLinCool/VoiceBank-DEMAND-16k","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":53,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-enhancement-on-demand","task":"Speech Enhancement","dataset_variant":"VoiceBank + DEMAND","rows":42,"metrics":["PESQ (wb)","CBAK","COVL","CSIG","STOI","ESTOI","SSNR","SI-SDR","Para. (M)"],"first_row_in_archive_order":{"model":"ROSE-CD(PESQ)","paper":"/paper/robust-one-step-speech-enhancement-via-1","metrics":{"CBAK":"3.37","COVL":"4.30","CSIG":"4.63","ESTOI":"0.83","PESQ (wb)":"3.99","Para. (M)":"65","SI-SDR":"0.40","SSNR":"0.927","STOI":"92.6"},"code_links":[{"title":"LiangXu123/Robust-One-step-Speech-Enhancement-via-Consistency-Distillation-ROSE-CD-","url":"https://github.com/LiangXu123/Robust-One-step-Speech-Enhancement-via-Consistency-Distillation-ROSE-CD-"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/robust-one-step-speech-enhancement-via-1","title":"Robust One-step Speech Enhancement via Consistency Distillation","date":"2025-07-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/primek-net-multi-scale-spectral-learning-via","title":"PrimeK-Net: Multi-scale Spectral Learning via Group Prime-Kernel Convolutional Neural Networks for Single Channel Speech Enhancement","date":"2025-02-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/let-ssms-be-convnets-state-space-modeling","title":"Let SSMs be ConvNets: State-space Modeling with Optimal Tensor Contractions","date":"2025-01-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/xlstm-senet-xlstm-for-single-channel-speech","title":"xLSTM-SENet: xLSTM for Single-Channel Speech Enhancement","date":"2025-01-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mamba-seunet-mamba-unet-for-monaural-speech","title":"Mamba-SEUNet: Mamba UNet for Monaural Speech Enhancement","date":"2024-12-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dense-tsnet-dense-connected-two-stage","title":"Dense-TSNet: Dense Connected Two-Stage Structure for Ultra-Lightweight Speech Enhancement","date":"2024-09-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/investigating-training-objectives-for","title":"Investigating Training Objectives for Generative Speech Enhancement","date":"2024-09-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/raw-speech-enhancement-with-deep-state-space","title":"aTENNuate: Optimized Real-time Speech Enhancement with Deep SSMs on Raw Audio","date":"2024-09-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/the-pesqetarian-on-the-relevance-of-goodhart","title":"The PESQetarian: On the Relevance of Goodhart's Law for Speech Enhancement","date":"2024-06-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/an-investigation-of-incorporating-mamba-for","title":"An Investigation of Incorporating Mamba for Speech Enhancement","date":"2024-05-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":13,"samples_unverified":1,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fspen-an-ultra-lightweight-network-for-real","title":"FSPEN: AN ULTRA-LIGHTWEIGHT NETWORK FOR REAL TIME SPEECH ENAHNCMENT","date":"2024-04-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/an-analysis-of-the-variance-of-diffusion","title":"An Analysis of the Variance of Diffusion-based Speech Enhancement","date":"2024-02-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/rose-a-recognition-oriented-speech","title":"ROSE: A Recognition-Oriented Speech Enhancement Framework in Air Traffic Control Using Multi-Objective Learning","date":"2023-12-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/explicit-estimation-of-magnitude-and-phase","title":"Explicit Estimation of Magnitude and Phase Spectra in Parallel for High-Quality Speech Enhancement","date":"2023-08-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":13,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/metricgan-okd-multi-metric-optimization-of","title":"MetricGAN-OKD: Multi-Metric Optimization of MetricGAN via Online Knowledge Distillation for Speech Enhancement","date":"2023-07-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/deepfilternet-perceptually-motivated-real","title":"DeepFilterNet: Perceptually Motivated Real-Time Speech Enhancement","date":"2023-05-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/d2net-a-denoising-and-dereverberation-network","title":"D²Net: A Denoising and Dereverberation Network Based on Two-branch Encoder and Dual-path Transformer","date":"2022-11-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scp-gan-self-correcting-discriminator","title":"SCP-GAN: Self-Correcting Discriminator Optimization for Training Consistency Preserving Metric GAN on Speech Enhancement Tasks","date":"2022-10-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cmgan-conformer-based-metric-gan-for-monaural","title":"CMGAN: Conformer-Based Metric-GAN for Monaural Speech Enhancement","date":"2022-09-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/multi-view-attention-transfer-for-efficient","title":"Multi-View Attention Transfer for Efficient Speech Enhancement","date":"2022-08-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/speech-enhancement-and-dereverberation-with","title":"Speech Enhancement and Dereverberation with Diffusion-based Generative Models","date":"2022-08-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/boosting-self-supervised-embeddings-for","title":"Boosting Self-Supervised Embeddings for Speech Enhancement","date":"2022-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/perceptual-contrast-stretching-on-target","title":"Perceptual Contrast Stretching on Target Feature for Speech Enhancement","date":"2022-03-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/manner-multi-view-attention-network-for-noise","title":"MANNER: Multi-view Attention Network for Noise Erasure","date":"2022-03-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/metricgan-an-improved-version-of-metricgan","title":"MetricGAN+: An Improved Version of MetricGAN for Speech Enhancement","date":"2021-04-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-modulation-domain-loss-for-neural-network-1","title":"A Modulation-Domain Loss for Neural-Network-based Real-time Speech Enhancement","date":"2021-02-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/monaural-speech-enhancement-with-complex","title":"Monaural Speech Enhancement with Complex Convolutional Block Attention Module and Joint Time Frequency Losses","date":"2021-02-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/improving-perceptual-quality-by-phone","title":"Improving Perceptual Quality by Phone-Fortified Perceptual Loss using Wasserstein Distance for Speech Enhancement","date":"2020-10-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/perceptual-loss-based-speech-denoising-with","title":"Perceptual Loss based Speech Denoising with an ensemble of Audio Pattern Recognition and Self-Supervised Models","date":"2020-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/real-time-speech-enhancement-in-the-waveform","title":"Real Time Speech Enhancement in the Waveform Domain","date":"2020-06-23","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-residual-dense-lattice-network-for","title":"Deep Residual-Dense Lattice Network for Speech Enhancement","date":"2020-02-27","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/metricgan-generative-adversarial-networks","title":"MetricGAN: Generative Adversarial Networks based Black-box Metric Scores Optimization for Speech Enhancement","date":"2019-05-13","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/end-to-end-speech-enhancement-based-on","title":"End-to-end speech enhancement based on discrete cosine transform","date":null,"rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":49,"samples_ran":29,"samples_unverified":20,"pointer_only_for_licence":16,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}