{"url":"/task/audio-source-separation","name":"Audio Source Separation","slug":"audio-source-separation","description_markdown":"**Audio Source Separation** is the process of separating a mixture (e.g. a pop band recording) into isolated sounds from individual sources (e.g. just the lead vocals).\r\n\r\n\r\n<span class=\"description-source\">Source: [Model selection for deep audio source separation via clustering analysis ](https://arxiv.org/abs/1910.12626)</span>","categories":[{"name":"Audio","url":"/area/audio"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":112,"papers_with_code":54,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":15,"subtasks":3,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/audio-source-separation-on-audioset","slug":"audio-source-separation-on-audioset","dataset":"AudioSet","dataset_url":"/dataset/audioset","rows_in_archive":2,"metrics":["SDR","SAR","SIR"],"first_row_in_archive_order":{"model":"ST-SED-SEP","paper_title":"Zero-shot Audio Source Separation through Query-based Learning from Weakly-labeled Data","paper_url":"/paper/zero-shot-audio-source-separation-through","paper_date":"2021-12-15","arxiv_id":"2112.07891","code_links":[{"title":"RetroCirce/Zero_Shot_Audio_Source_Separation","url":"https://github.com/RetroCirce/Zero_Shot_Audio_Source_Separation"}],"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}}},{"leaderboard":"/sota/audio-source-separation-on-music-multi-source","slug":"audio-source-separation-on-music-multi-source","dataset":"MUSIC (multi-source)","dataset_url":null,"rows_in_archive":1,"metrics":["SAR","SIR"],"first_row_in_archive_order":{"model":"Co-Separation","paper_title":"Co-Separating Sounds of Visual Objects","paper_url":"/paper/co-separating-sounds-of-visual-objects","paper_date":"2019-04-16","arxiv_id":"1904.07750","code_links":[{"title":"rhgao/co-separation","url":"https://github.com/rhgao/co-separation"},{"title":"manhnguyen1998/co_separation_encoder_decoder","url":"https://github.com/manhnguyen1998/co_separation_encoder_decoder"},{"title":"YashNita/Co-Separating-Sound-Object-","url":"https://github.com/YashNita/Co-Separating-Sound-Object-"}],"syntology":null}}],"datasets":[{"url":"/dataset/audioset","name":"AudioSet","full_name":"","num_papers_in_archive":744},{"url":"/dataset/wsj0-2mix-1","name":"WSJ0-2mix","full_name":"","num_papers_in_archive":159},{"url":"/dataset/librimix","name":"LibriMix","full_name":"","num_papers_in_archive":122},{"url":"/dataset/wham","name":"WHAM!","full_name":"WSJ0 Hipster Ambient Mixtures","num_papers_in_archive":114},{"url":"/dataset/musdb18","name":"MUSDB18","full_name":"","num_papers_in_archive":106},{"url":"/dataset/whamr","name":"WHAMR!","full_name":"WHAM! with synthetic reverberated sources","num_papers_in_archive":57},{"url":"/dataset/deep-noise-suppression-2020","name":"DNS Challenge","full_name":"Deep Noise Suppression Challenge","num_papers_in_archive":48},{"url":"/dataset/avspeech","name":"AVSpeech","full_name":"","num_papers_in_archive":42},{"url":"/dataset/fuss","name":"FUSS","full_name":"Free Universal Sound Separation","num_papers_in_archive":16},{"url":"/dataset/sms-wsj","name":"SMS-WSJ","full_name":"Spatialized Multi-Speaker Wall Street Journal","num_papers_in_archive":14},{"url":"/dataset/openmic-2018","name":"OpenMIC-2018","full_name":null,"num_papers_in_archive":8},{"url":"/dataset/medleyvox","name":"MedleyVox","full_name":"","num_papers_in_archive":2},{"url":"/dataset/jacappella","name":"jaCappella","full_name":"","num_papers_in_archive":1},{"url":"/dataset/kinect-wsj","name":"Kinect-WSJ","full_name":"","num_papers_in_archive":1},{"url":"/dataset/cadenza-woodwind","name":"Cadenza Woodwind","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/directional-hearing","name":"Directional Hearing"},{"url":"/task/single-label-target-sound-extraction","name":"Single-Label Target Sound Extraction"},{"url":"/task/target-sound-extraction","name":"Target Sound Extraction"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":54,"tagged_in_all":112,"items":[{"url":"/paper/wave-u-net-a-multi-scale-neural-network-for","title":"Wave-U-Net: A Multi-Scale Neural Network for End-to-End Audio Source Separation","date":"2018-06-08","arxiv_id":"1806.03185","repositories_listed":10,"syntology":{"n":19,"n_ran":1,"n_unverified":18,"n_pointer_only":1}},{"url":"/paper/multi-scale-multi-band-densenets-for-audio","title":"Multi-scale Multi-band DenseNets for Audio Source Separation","date":"2017-06-29","arxiv_id":"1706.09588","repositories_listed":5,"syntology":null},{"url":"/paper/sudo-rm-rf-efficient-networks-for-universal","title":"Sudo rm -rf: Efficient Networks for Universal Audio Source Separation","date":"2020-07-14","arxiv_id":"2007.06833","repositories_listed":4,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/the-cocktail-fork-problem-three-stem-audio","title":"The Cocktail Fork Problem: Three-Stem Audio Separation for Real-World Soundtracks","date":"2021-10-19","arxiv_id":"2110.09958","repositories_listed":3,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/compute-and-memory-efficient-universal-sound","title":"Compute and memory efficient universal sound source separation","date":"2021-03-03","arxiv_id":"2103.02644","repositories_listed":3,"syntology":null},{"url":"/paper/co-separating-sounds-of-visual-objects","title":"Co-Separating Sounds of Visual Objects","date":"2019-04-16","arxiv_id":"1904.07750","repositories_listed":3,"syntology":null},{"url":"/paper/improved-speech-enhancement-with-the-wave-u","title":"Improved Speech Enhancement with the Wave-U-Net","date":"2018-11-27","arxiv_id":"1811.11307","repositories_listed":3,"syntology":null},{"url":"/paper/adversarial-semi-supervised-audio-source","title":"Adversarial Semi-Supervised Audio Source Separation applied to Singing Voice Extraction","date":"2017-10-31","arxiv_id":"1711.00048","repositories_listed":3,"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/unsupervised-audio-source-separation-using-1","title":"Unsupervised Music Source Separation Using Differentiable Parametric Source Models","date":"2022-01-24","arxiv_id":"2201.09592","repositories_listed":2,"syntology":null},{"url":"/paper/directional-sparse-filtering-using-weighted","title":"Directional Sparse Filtering using Weighted Lehmer Mean for Blind Separation of Unbalanced Speech Mixtures","date":"2021-01-30","arxiv_id":"2102.00196","repositories_listed":2,"syntology":null},{"url":"/paper/conditioned-u-net-introducing-a-control","title":"Conditioned-U-Net: Introducing a Control Mechanism in the U-Net for Multiple Source Separations","date":"2019-07-02","arxiv_id":"1907.01277","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/learning-to-separate-object-sounds-by","title":"Learning to Separate Object Sounds by Watching Unlabeled Video","date":"2018-04-05","arxiv_id":"1804.01665","repositories_listed":2,"syntology":null},{"url":"/paper/towards-reliable-objective-evaluation-metrics","title":"Towards Reliable Objective Evaluation Metrics for Generative Singing Voice Separation Models","date":"2025-07-15","arxiv_id":"2507.11427","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-multi-step-audio-source","title":"Training-Free Multi-Step Audio Source Separation","date":"2025-05-26","arxiv_id":"2505.19534","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-llm-and-text-queried-separation","title":"Leveraging LLM and Text-Queried Separation for Noise-Robust Sound Event Detection","date":"2024-11-02","arxiv_id":"2411.01174","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-text-queried-sound-event-detection","title":"Exploring Text-Queried Sound Event Detection with Audio Source Separation","date":"2024-09-20","arxiv_id":"2409.13292","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/unsupervised-composable-representations-for","title":"Unsupervised Composable Representations for Audio","date":"2024-08-19","arxiv_id":"2408.09792","repositories_listed":1,"syntology":null},{"url":"/paper/facing-the-music-tackling-singing-voice","title":"Facing the Music: Tackling Singing Voice Separation in Cinematic Audio Source Separation","date":"2024-08-07","arxiv_id":"2408.03588","repositories_listed":1,"syntology":null},{"url":"/paper/remastering-divide-and-remaster-a-cinematic","title":"Remastering Divide and Remaster: A Cinematic Audio Source Separation Dataset with Multilingual Support","date":"2024-07-09","arxiv_id":"2407.07275","repositories_listed":1,"syntology":null},{"url":"/paper/a-stem-agnostic-single-decoder-system-for","title":"A Stem-Agnostic Single-Decoder System for Music Source Separation Beyond Four Stems","date":"2024-06-26","arxiv_id":"2406.18747","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/spectral-mapping-of-singing-voices-u-net","title":"Spectral Mapping of Singing Voices: U-Net-Assisted Vocal Segmentation","date":"2024-05-30","arxiv_id":"2405.20059","repositories_listed":1,"syntology":null},{"url":"/paper/a-generalized-bandsplit-neural-network-for","title":"A Generalized Bandsplit Neural Network for Cinematic Audio Source Separation","date":"2023-09-05","arxiv_id":"2309.02539","repositories_listed":1,"syntology":null},{"url":"/paper/separate-anything-you-describe","title":"Separate Anything You Describe","date":"2023-08-09","arxiv_id":"2308.05037","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/learning-audio-visual-dynamics-using-scene","title":"Learning Audio-Visual Dynamics Using Scene Graphs for Audio Source Separation","date":"2022-10-29","arxiv_id":"2210.16472","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/deep-audio-waveform-prior","title":"Deep Audio Waveform Prior","date":"2022-07-21","arxiv_id":"2207.10441","repositories_listed":1,"syntology":null},{"url":"/paper/separate-what-you-describe-language-queried","title":"Separate What You Describe: Language-Queried Audio Source Separation","date":"2022-03-28","arxiv_id":"2203.15147","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/zero-shot-audio-source-separation-through","title":"Zero-shot Audio Source Separation through Query-based Learning from Weakly-labeled Data","date":"2021-12-15","arxiv_id":"2112.07891","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/hybrid-neural-networks-for-on-device","title":"Hybrid Neural Networks for On-device Directional Hearing","date":"2021-12-11","arxiv_id":"2112.05893","repositories_listed":1,"syntology":null},{"url":"/paper/transfer-learning-with-jukebox-for-music","title":"Transfer Learning with Jukebox for Music Source Separation","date":"2021-11-28","arxiv_id":"2111.14200","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-source-separation-by-steering","title":"Unsupervised Source Separation By Steering Pretrained Music Models","date":"2021-10-25","arxiv_id":"2110.13071","repositories_listed":1,"syntology":null}],"syntology_records":11,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}