{"url":"/task/music-source-separation","name":"Music Source Separation","slug":"music-source-separation","description_markdown":"Music source separation is the task of decomposing music into its constitutive components, e. g., yielding separated stems for the vocals, bass, and drums.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [SigSep](https://github.com/sigsep ) )</span>","categories":[{"name":"Music","url":"/area/music"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":107,"papers_with_code":58,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/music-source-separation-on-musdb18","slug":"music-source-separation-on-musdb18","dataset":"MUSDB18","dataset_url":"/dataset/musdb18","rows_in_archive":27,"metrics":["SDR (avg)","SDR (vocals)","SDR (drums)","SDR (bass)","SDR (other)"],"first_row_in_archive_order":{"model":"Sparse HT Demucs (fine tuned)","paper_title":"Hybrid Transformers for Music Source Separation","paper_url":"/paper/hybrid-transformers-for-music-source","paper_date":"2022-11-15","arxiv_id":"2211.08553","code_links":[{"title":"facebookresearch/demucs","url":"https://github.com/facebookresearch/demucs"},{"title":"zhaozhipeng1997/demucs_ascend910_pytorch","url":"https://github.com/zhaozhipeng1997/demucs_ascend910_pytorch"}],"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":10}}},{"leaderboard":"/sota/music-source-separation-on-musdb18-hq","slug":"music-source-separation-on-musdb18-hq","dataset":"MUSDB18-HQ","dataset_url":"/dataset/musdb18-hq","rows_in_archive":14,"metrics":["SDR (avg)","SDR (bass)","SDR (drums)","SDR (others)","SDR (vocals)"],"first_row_in_archive_order":{"model":"BS-RoFormer (L=12, OA)","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/music-source-separation-on-slakh2100","slug":"music-source-separation-on-slakh2100","dataset":"Slakh2100","dataset_url":"/dataset/slakh2100","rows_in_archive":2,"metrics":["SDR (bass)","SDR (drums)","SI-SDRi (Bass)","Si-SDRi (Drums)","Si-SDRi (Guitar)","Si-SDRi (Piano)"],"first_row_in_archive_order":{"model":"LQ-VAE + Scalable Transformer","paper_title":"Unsupervised Source Separation via Bayesian Inference in the Latent Domain","paper_url":"/paper/unsupervised-source-separation-via-bayesian","paper_date":"2021-10-11","arxiv_id":"2110.05313","code_links":[{"title":"michelemancusi/LQVAE-separation","url":"https://github.com/michelemancusi/LQVAE-separation"}],"syntology":null}}],"datasets":[{"url":"/dataset/musdb18","name":"MUSDB18","full_name":"","num_papers_in_archive":106},{"url":"/dataset/medleydb","name":"MedleyDB","full_name":"","num_papers_in_archive":47},{"url":"/dataset/slakh2100","name":"Slakh2100","full_name":"Synthesized Lakh Dataset","num_papers_in_archive":38},{"url":"/dataset/mir-1k","name":"MIR-1K","full_name":"","num_papers_in_archive":21},{"url":"/dataset/musdb18-hq","name":"MUSDB18-HQ","full_name":null,"num_papers_in_archive":15},{"url":"/dataset/cocochorales","name":"CocoChorales","full_name":"","num_papers_in_archive":7},{"url":"/dataset/musescore","name":"MuseScore","full_name":"","num_papers_in_archive":2},{"url":"/dataset/synthsod","name":"SynthSOD","full_name":"","num_papers_in_archive":2},{"url":"/dataset/cadenza-woodwind","name":"Cadenza Woodwind","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[{"url":"/task/2d-classification","name":"2D Classification"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":58,"tagged_in_all":107,"items":[{"url":"/paper/tasnet-surpassing-ideal-time-frequency","title":"Conv-TasNet: Surpassing Ideal Time-Frequency Magnitude Masking for Speech Separation","date":"2018-09-20","arxiv_id":"1809.07454","repositories_listed":17,"syntology":{"n":36,"n_ran":7,"n_unverified":29,"n_pointer_only":16}},{"url":"/paper/wave-u-net-a-multi-scale-neural-network-for","title":"Wave-U-Net: A Multi-Scale Neural Network for End-to-End Audio Source Separation","date":"2018-06-08","arxiv_id":"1806.03185","repositories_listed":10,"syntology":{"n":19,"n_ran":1,"n_unverified":18,"n_pointer_only":1}},{"url":"/paper/all-for-one-and-one-for-all-improving-music","title":"All for One and One for All: Improving Music Separation by Bridging Networks","date":"2020-10-08","arxiv_id":"2010.04228","repositories_listed":5,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/multi-scale-multi-band-densenets-for-audio","title":"Multi-scale Multi-band DenseNets for Audio Source Separation","date":"2017-06-29","arxiv_id":"1706.09588","repositories_listed":5,"syntology":null},{"url":"/paper/scnet-sparse-compression-network-for-music","title":"SCNet: Sparse Compression Network for Music Source Separation","date":"2024-01-24","arxiv_id":"2401.13276","repositories_listed":3,"syntology":null},{"url":"/paper/music-source-separation-with-band-split-rnn","title":"Music Source Separation with Band-split RNN","date":"2022-09-30","arxiv_id":"2209.15174","repositories_listed":3,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/spleeter-a-fast-and-state-of-the-art-music","title":"Spleeter: A Fast And State-of-the Art Music Source Separation Tool With Pre-trained Models","date":"2019-11-04","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/adversarial-semi-supervised-audio-source","title":"Adversarial Semi-Supervised Audio Source Separation applied to Singing Voice Extraction","date":"2017-10-31","arxiv_id":"1711.00048","repositories_listed":3,"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/learned-compression-for-compressed-learning","title":"Learned Compression for Compressed Learning","date":"2024-12-12","arxiv_id":"2412.09405","repositories_listed":2,"syntology":null},{"url":"/paper/synthsod-developing-an-heterogeneous-dataset","title":"SynthSOD: Developing an Heterogeneous Dataset for Orchestra Music Source Separation","date":"2024-09-17","arxiv_id":"2409.10995","repositories_listed":2,"syntology":null},{"url":"/paper/pre-training-music-classification-models-via","title":"Pre-training Music Classification Models via Music Source Separation","date":"2023-10-24","arxiv_id":"2310.15845","repositories_listed":2,"syntology":null},{"url":"/paper/music-source-separation-based-on-a","title":"Music Source Separation Based on a Lightweight Deep Learning Framework (DTTNET: DUAL-PATH TFC-TDF UNET)","date":"2023-09-15","arxiv_id":"2309.08684","repositories_listed":2,"syntology":null},{"url":"/paper/the-sound-demixing-challenge-2023-unicode-1","title":"The Sound Demixing Challenge 2023 $\\unicode{x2013}$ Music Demixing Track","date":"2023-08-14","arxiv_id":"2308.06979","repositories_listed":2,"syntology":null},{"url":"/paper/hybrid-transformers-for-music-source","title":"Hybrid Transformers for Music Source Separation","date":"2022-11-15","arxiv_id":"2211.08553","repositories_listed":2,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":10}},{"url":"/paper/unsupervised-audio-source-separation-using-1","title":"Unsupervised Music Source Separation Using Differentiable Parametric Source Models","date":"2022-01-24","arxiv_id":"2201.09592","repositories_listed":2,"syntology":null},{"url":"/paper/a-cappella-audio-visual-singing-voice","title":"A cappella: Audio-visual Singing Voice Separation","date":"2021-04-20","arxiv_id":"2104.09946","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/multi-task-u-net-for-music-source-separation","title":"Multi-channel U-Net for Music Source Separation","date":"2020-03-23","arxiv_id":"2003.10414","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-music-source-separation-is-it","title":"End-to-end music source separation: is it possible in the waveform domain?","date":"2018-10-29","arxiv_id":"1810.12187","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/primal-dual-algorithms-for-non-negative","title":"Primal-Dual Algorithms for Non-negative Matrix Factorization with the Kullback-Leibler Divergence","date":"2014-12-04","arxiv_id":"1412.1788","repositories_listed":2,"syntology":null},{"url":"/paper/music-source-restoration","title":"Music Source Restoration","date":"2025-05-27","arxiv_id":"2505.21827","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-multi-step-audio-source","title":"Training-Free Multi-Step Audio Source Separation","date":"2025-05-26","arxiv_id":"2505.19534","repositories_listed":1,"syntology":null},{"url":"/paper/score-informed-music-source-separation","title":"Score-informed Music Source Separation: Improving Synthetic-to-real Generalization in Classical Music","date":"2025-03-10","arxiv_id":"2503.07352","repositories_listed":1,"syntology":null},{"url":"/paper/a-stem-agnostic-single-decoder-system-for","title":"A Stem-Agnostic Single-Decoder System for Music Source Separation Beyond Four Stems","date":"2024-06-26","arxiv_id":"2406.18747","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/a-fully-differentiable-model-for-unsupervised","title":"A fully differentiable model for unsupervised singing voice separation","date":"2024-01-30","arxiv_id":"2401.16837","repositories_listed":1,"syntology":null},{"url":"/paper/machine-perceptual-quality-evaluating-the","title":"Machine Perceptual Quality: Evaluating the Impact of Severe Lossy Compression on Audio and Image Models","date":"2024-01-15","arxiv_id":"2401.07957","repositories_listed":1,"syntology":null},{"url":"/paper/sound-demixing-challenge-2023-music-demixing","title":"Sound Demixing Challenge 2023 Music Demixing Track Technical Report: TFC-TDF-UNet v3","date":"2023-06-15","arxiv_id":"2306.09382","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-spatial-distortion-in","title":"Quantifying Spatial Audio Quality Impairment","date":"2023-06-13","arxiv_id":"2306.08053","repositories_listed":1,"syntology":null},{"url":"/paper/the-whole-is-greater-than-the-sum-of-its-3","title":"The Whole Is Greater than the Sum of Its Parts: Improving Music Source Separation by Bridging Network","date":"2023-05-13","arxiv_id":"2305.07855","repositories_listed":1,"syntology":null},{"url":"/paper/medleyvox-an-evaluation-dataset-for-multiple","title":"MedleyVox: An Evaluation Dataset for Multiple Singing Voices Separation","date":"2022-11-14","arxiv_id":"2211.07302","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-short-time-discrete-cosine","title":"An Efficient Short-Time Discrete Cosine Transform and Attentive MultiResUNet Framework for Music Source Separation","date":"2022-11-14","arxiv_id":null,"repositories_listed":1,"syntology":null}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}