{"url":"/dataset/whamr","name":"WHAMR!","full_name":"WHAM! with synthetic reverberated sources","description_markdown":"**WHAMR!** is a dataset for noisy and reverberant speech separation. It extends [WHAM!](/dataset/wham) by introducing synthetic reverberation to the\r\nspeech sources in addition to the existing noise. Room impulse responses were generated and convolved using `pyroomacoustics`. Reverberation times were chosen to approximate domestic and classroom environments (expected to be similar to the restaurants and coffee shops where the WHAM! noise was collected), and\r\nfurther classified as high, medium, and low reverberation based on a\r\nqualitative assessment of the mixture’s noise recording.","description_withheld":null,"homepage":"https://wham.whisper.ai/","introduced_date":"2019-10-22","introduced_date_note":null,"introduced_by":{"paper":null,"title":"WHAMR!: Noisy and Reverberant Single-Channel Speech Separation","first_author":null,"url":null},"license":null,"modalities":[{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Enhancement","url":"/task/speech-enhancement","datasets_with_task":"/datasets/task/speech-enhancement"},{"name":"Speech Separation","url":"/task/speech-separation","datasets_with_task":"/datasets/task/speech-separation"},{"name":"Audio Source Separation","url":"/task/audio-source-separation","datasets_with_task":"/datasets/task/audio-source-separation"},{"name":"Speech Dereverberation","url":"/task/speech-dereverberation","datasets_with_task":"/datasets/task/speech-dereverberation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["WHAMR!"],"data_loaders":[],"num_papers_in_archive":57,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-separation-on-whamr","task":"Speech Separation","dataset_variant":"WHAMR!","rows":18,"metrics":["SI-SDRi","MACs (G)","Number of parameters (M)","SDRi"],"first_row_in_archive_order":{"model":"TF-Locoformer (M)","paper":"/paper/tf-locoformer-transformer-with-local-modeling","metrics":{"Number of parameters (M)":"15","SDRi":"16.9","SI-SDRi":"18.5"},"code_links":[{"title":"merlresearch/tf-locoformer","url":"https://github.com/merlresearch/tf-locoformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-enhancement-on-whamr","task":"Speech Enhancement","dataset_variant":"WHAMR!","rows":4,"metrics":["PESQ","SI-SDR","ΔPESQ","SI-SNR","SDR"],"first_row_in_archive_order":{"model":"SepFormer","paper":"/paper/on-using-transformers-for-speech-separation","metrics":{"PESQ":"2.84","SDR":"12.29","SI-SNR":"10.58"},"code_links":[{"title":"speechbrain/speechbrain","url":"https://github.com/speechbrain/speechbrain"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-dereverberation-on-whamr","task":"Speech Dereverberation","dataset_variant":"WHAMR!","rows":3,"metrics":["PESQ","SI-SDR","ESTOI","SRMR","SI-SDRi"],"first_row_in_archive_order":{"model":"WD-TCN","paper":"/paper/utterance-weighted-multi-dilation-temporal","metrics":{"ESTOI":"93.5","PESQ":"3.5","SI-SDR":"12.26","SRMR":"8.8"},"code_links":[{"title":"jwr1995/wd-tcn","url":"https://github.com/jwr1995/wd-tcn"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/tf-locoformer-transformer-with-local-modeling","title":"TF-Locoformer: Transformer with Local Modeling by Convolution for Speech Separation and Enhancement","date":"2024-08-06","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/separate-and-reconstruct-asymmetric-encoder","title":"Separate and Reconstruct: Asymmetric Encoder-Decoder for Speech Separation","date":"2024-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/on-time-domain-conformer-models-for-monaural","title":"On Time Domain Conformer Models for Monaural Speech Separation in Noisy Reverberant Acoustic Environments","date":"2023-10-09","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/mossformer-pushing-the-performance-limit-of","title":"MossFormer: Pushing the Performance Limit of Monaural Speech Separation using Gated Single-Head Transformer with Convolution-Augmented Joint Self-Attentions","date":"2023-02-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deformable-temporal-convolutional-networks","title":"Deformable Temporal Convolutional Networks for Monaural Noisy Reverberant Speech Separation","date":"2022-10-27","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/utterance-weighted-multi-dilation-temporal","title":"Utterance Weighted Multi-Dilation Temporal Convolutional Networks for Monaural Speech Dereverberation","date":"2022-05-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/receptive-field-analysis-of-temporal","title":"Receptive Field Analysis of Temporal Convolutional Networks for Monaural Speech Dereverberation","date":"2022-04-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/on-using-transformers-for-speech-separation","title":"Exploring Self-Attention Mechanisms for Speech Separation","date":"2022-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stepwise-refining-speech-separation-network","title":"Stepwise-Refining Speech Separation Network via Fine-Grained Encoding in High-order Latent Domain","date":"2021-10-10","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/compute-and-memory-efficient-universal-sound","title":"Compute and memory efficient universal sound source separation","date":"2021-03-03","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/sudo-rm-rf-efficient-networks-for-universal","title":"Sudo rm -rf: Efficient Networks for Universal Audio Source Separation","date":"2020-07-14","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/phase-aware-single-stage-speech-denoising-and-1","title":"Phase-aware Single-stage Speech Denoising and Dereverberation with U-Net","date":"2020-06-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/voice-separation-with-an-unknown-number-of","title":"Voice Separation with an Unknown Number of Multiple Speakers","date":"2020-02-29","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":10,"samples_unverified":3,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/wavesplit-end-to-end-speech-separation-by","title":"Wavesplit: End-to-End Speech Separation by Speaker Clustering","date":"2020-02-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/wham-extending-speech-separation-to-noisy","title":"WHAM!: Extending Speech Separation to Noisy Environments","date":"2019-07-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mossformer2-combining-transformer-and-rnn-1","title":"MossFormer2: Combining Transformer and RNN-Free Recurrent Network for Enhanced Time-Domain Monaural Speech Separation","date":null,"rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dual-signal-transformation-lstm-network-for","title":"Dual-Signal Transformation LSTM Network for Real-Time Noise Suppression","date":null,"rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":32,"samples_ran":16,"samples_unverified":16,"pointer_only_for_licence":10,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}