{"url":"/dataset/wsj0-2mix-1","name":"WSJ0-2mix","full_name":null,"description_markdown":"**WSJ0-2mix** is a speech recognition corpus of speech mixtures using utterances from the Wall Street Journal (WSJ0) corpus.\r\n\r\nSource: [Deep clustering: Discriminative embeddings for segmentation and separation](/paper/deep-clustering-discriminative-embeddings-for)","description_withheld":null,"homepage":"https://www.merl.com/demos/deep-clustering","introduced_date":"2015-08-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/deep-clustering-discriminative-embeddings-for","title":"Deep clustering: Discriminative embeddings for segmentation and separation","first_author":"John R. Hershey","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Separation","url":"/task/speech-separation","datasets_with_task":"/datasets/task/speech-separation"},{"name":"Audio Source Separation","url":"/task/audio-source-separation","datasets_with_task":"/datasets/task/audio-source-separation"},{"name":"Adversarial Attack","url":"/task/adversarial-attack","datasets_with_task":"/datasets/task/adversarial-attack"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["WSJ0-2mix","WSJ0-2mix-16k"],"data_loaders":[{"repo":"https://github.com/JusperLee/Conv-TasNet","url":"https://github.com/JusperLee/Conv-TasNet","frameworks":["pytorch"]}],"num_papers_in_archive":159,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-separation-on-wsj0-2mix","task":"Speech Separation","dataset_variant":"WSJ0-2mix","rows":40,"metrics":["SI-SDRi","SDRi","Number of parameters (M)","MACs (G)"],"first_row_in_archive_order":{"model":"TF-Locoformer (L) + DM","paper":"/paper/tf-locoformer-transformer-with-local-modeling","metrics":{"Number of parameters (M)":"22.5","SDRi":"25.2","SI-SDRi":"25.1"},"code_links":[{"title":"merlresearch/tf-locoformer","url":"https://github.com/merlresearch/tf-locoformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/adversarial-attack-on-wsj0-2mix","task":"Adversarial Attack","dataset_variant":"WSJ0-2mix","rows":1,"metrics":["SDR"],"first_row_in_archive_order":{"model":"ConvTasnet and Dual Path Transformers","paper":"/paper/harmonicity-plays-a-critical-role-in-dnn","metrics":{"SDR":"0.70dB"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-separation-on-wsj0-2mix-16k","task":"Speech Separation","dataset_variant":"WSJ0-2mix-16k","rows":1,"metrics":["SI-SDRi"],"first_row_in_archive_order":{"model":"MossFormer2","paper":"/paper/mossformer-pushing-the-performance-limit-of","metrics":{"SI-SDRi":"20.5"},"code_links":[{"title":"modelscope/ClearerVoice-Studio","url":"https://github.com/modelscope/ClearerVoice-Studio"},{"title":"alibabasglab/mossformer","url":"https://github.com/alibabasglab/mossformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/sepmamba-state-space-models-for-speaker","title":"SepMamba: State-space models for speaker separation using Mamba","date":"2024-10-28","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/tf-locoformer-transformer-with-local-modeling","title":"TF-Locoformer: Transformer with Local Modeling by Convolution for Speech Separation and Enhancement","date":"2024-08-06","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/separate-and-reconstruct-asymmetric-encoder","title":"Separate and Reconstruct: Asymmetric Encoder-Decoder for Speech Separation","date":"2024-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/boosting-unknown-number-speaker-separation","title":"Boosting Unknown-number Speaker Separation with Transformer Decoder-based Attractor","date":"2024-01-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/on-time-domain-conformer-models-for-monaural","title":"On Time Domain Conformer Models for Monaural Speech Separation in Noisy Reverberant Acoustic Environments","date":"2023-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/spgm-prioritizing-local-features-for-enhanced","title":"SPGM: Prioritizing Local Features for enhanced speech separation performance","date":"2023-09-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/mossformer-pushing-the-performance-limit-of","title":"MossFormer: Pushing the Performance Limit of Monaural Speech Separation using Gated Single-Head Transformer with Convolution-Augmented Joint Self-Attentions","date":"2023-02-23","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/separate-and-diffuse-using-a-pretrained","title":"Separate And Diffuse: Using a Pretrained Diffusion Model for Improving Source Separation","date":"2023-01-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deformable-temporal-convolutional-networks","title":"Deformable Temporal Convolutional Networks for Monaural Noisy Reverberant Speech Separation","date":"2022-10-27","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/sepit-approaching-a-single-channel-speech","title":"SepIt: Approaching a Single Channel Speech Separation Bound","date":"2022-05-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/harmonicity-plays-a-critical-role-in-dnn","title":"Harmonicity Plays a Critical Role in DNN Based Versus in Biologically-Inspired Monaural Speech Segregation Systems","date":"2022-03-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/compute-and-memory-efficient-universal-sound","title":"Compute and memory efficient universal sound source separation","date":"2021-03-03","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/sandglasset-a-light-multi-granularity-self","title":"Sandglasset: A Light Multi-Granularity Self-attentive Network For Time-Domain Speech Separation","date":"2021-03-01","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/effective-low-cost-time-domain-audio","title":"Effective Low-Cost Time-Domain Audio Separation Using Globally Attentive Locally Recurrent Networks","date":"2021-01-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-pre-training-reduces-label","title":"Stabilizing Label Assignment for Speech Separation by Self-supervised Pre-training","date":"2020-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/attention-is-all-you-need-in-speech","title":"Attention is All You Need in Speech Separation","date":"2020-10-25","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/sudo-rm-rf-efficient-networks-for-universal","title":"Sudo rm -rf: Efficient Networks for Universal Audio Source Separation","date":"2020-07-14","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/voice-separation-with-an-unknown-number-of","title":"Voice Separation with an Unknown Number of Multiple Speakers","date":"2020-02-29","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":10,"samples_unverified":3,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/wavesplit-end-to-end-speech-separation-by","title":"Wavesplit: End-to-End Speech Separation by Speaker Clustering","date":"2020-02-20","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/interrupted-and-cascaded-permutation","title":"Interrupted and cascaded permutation invariant training for speech separation","date":"2019-10-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/two-step-sound-source-separation-training-on","title":"Two-Step Sound Source Separation: Training on Learned Latent Targets","date":"2019-10-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/dual-path-rnn-efficient-long-sequence","title":"Dual-path RNN: efficient long sequence modeling for time-domain single-channel speech separation","date":"2019-10-14","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":4,"samples_unverified":18,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/divide-and-conquer-a-deep-casa-approach-to","title":"Divide and Conquer: A Deep CASA Approach to Talker-independent Monaural Speaker Separation","date":"2019-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improved-speech-separation-with-time-and","title":"Improved Speech Separation with Time-and-Frequency Cross-domain Joint Embedding and Clustering","date":"2019-04-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tasnet-surpassing-ideal-time-frequency","title":"Conv-TasNet: Surpassing Ideal Time-Frequency Magnitude Masking for Speech Separation","date":"2018-09-20","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":36,"samples_ran":7,"samples_unverified":29,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/real-time-single-channel-dereverberation-and","title":"Real-time Single-channel Dereverberation and Separation with Time-domainAudio Separation Network","date":"2018-09-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/alternative-objective-functions-for-deep","title":"Alternative Objective Functions for Deep Clustering","date":"2018-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tasnet-time-domain-audio-separation-network","title":"TasNet: time-domain audio separation network for real-time, single-channel speech separation","date":"2017-11-01","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/deep-clustering-discriminative-embeddings-for","title":"Deep clustering: Discriminative embeddings for segmentation and separation","date":"2015-08-18","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mossformer2-combining-transformer-and-rnn-1","title":"MossFormer2: Combining Transformer and RNN-Free Recurrent Network for Enhanced Time-Domain Monaural Speech Separation","date":null,"rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dual-path-transformer-network-direct-context-1","title":"Dual-Path Transformer Network: Direct Context-Aware Modeling for End-to-End Monaural Speech Separation","date":null,"rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":99,"samples_ran":34,"samples_unverified":65,"pointer_only_for_licence":33,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}