{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/multi-decoder-dprnn-high-accuracy-source","title":"Multi-Decoder DPRNN: High Accuracy Source Counting and Separation","arxiv_id":"2011.12022","date":"2020-11-24","proceeding":null,"authors":["Junzhe Zhu","Raymond Yeh","Mark Hasegawa-Johnson"],"abstract":"We propose an end-to-end trainable approach to single-channel speech separation with unknown number of speakers. Our approach extends the MulCat source separation backbone with additional output heads: a count-head to infer the number of speakers, and decoder-heads for reconstructing the original signals. Beyond the model, we also propose a metric on how to evaluate source separation with variable number of speakers. Specifically, we cleared up the issue on how to evaluate the quality when the ground-truth hasmore or less speakers than the ones predicted by the model. We evaluate our approach on the WSJ0-mix datasets, with mixtures up to five speakers. We demonstrate that our approach outperforms state-of-the-art in counting the number of speakers and remains competitive in quality of reconstructed signals.","url_abs":"https://arxiv.org/abs/2011.12022v2","url_pdf":"https://arxiv.org/pdf/2011.12022v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"multi-decoder-dprnn-high-accuracy-source","repo_url":"https://github.com/asteroid-team/asteroid/tree/master/egs/wsj0-mix-var/Multi-Decoder-DPRNN","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"multi-decoder-dprnn-high-accuracy-source","repo_url":"https://github.com/JunzheJosephZhu/Multi-Decoder-DPRNN","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"decoder","task_name":"Decoder"},{"task_slug":"speech-separation","task_name":"Speech Separation"},{"task_slug":"high","task_name":"Vocal Bursts Intensity Prediction"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/speech-separation-on-wsj0-4mix","task":"Speech Separation","dataset":"WSJ0-4mix","model":"Multi-Decoder DPRNN","rank_in_archive_order":5,"of":5,"metrics":{"SI-SDRi":"9.3"},"uses_additional_data":false},{"leaderboard":"/sota/speech-separation-on-wsj0-5mix","task":"Speech Separation","dataset":"WSJ0-5mix","model":"Multi-Decoder DPRNN","rank_in_archive_order":6,"of":6,"metrics":{"SI-SDRi":"5.9"},"uses_additional_data":false}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}