{"url":"/task/target-sound-extraction","name":"Target Sound Extraction","slug":"target-sound-extraction","description_markdown":"Target Sound Extraction is the task of extracting a sound corresponding to a given class from an audio mixture. The audio mixture may contain background noise with a relatively low amplitude compared to the foreground mixture components. The choice of the sound class is provided as input to the model in form of a string, integer, or a one-hot encoding of the sound class.","categories":[{"name":"Audio","url":"/area/audio"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":16,"papers_with_code":8,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":1,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/target-sound-extraction-on-audiocaps","slug":"target-sound-extraction-on-audiocaps","dataset":"AudioCaps","dataset_url":"/dataset/audiocaps","rows_in_archive":1,"metrics":["SDRi","SI-SDRi"],"first_row_in_archive_order":{"model":"CLAPSep","paper_title":"CLAPSep: Leveraging Contrastive Pre-trained Model for Multi-Modal Query-Conditioned Target Sound Extraction","paper_url":"/paper/clapsep-leveraging-contrastive-pre-trained","paper_date":"2024-02-27","arxiv_id":"2402.17455","code_links":[{"title":"aisaka0v0/clapsep","url":"https://github.com/aisaka0v0/clapsep"}],"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/target-sound-extraction-on-audioset","slug":"target-sound-extraction-on-audioset","dataset":"AudioSet","dataset_url":"/dataset/audioset","rows_in_archive":1,"metrics":["SDRi","SI-SDRi"],"first_row_in_archive_order":{"model":"CLAPSep","paper_title":"CLAPSep: Leveraging Contrastive Pre-trained Model for Multi-Modal Query-Conditioned Target Sound Extraction","paper_url":"/paper/clapsep-leveraging-contrastive-pre-trained","paper_date":"2024-02-27","arxiv_id":"2402.17455","code_links":[{"title":"aisaka0v0/clapsep","url":"https://github.com/aisaka0v0/clapsep"}],"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/target-sound-extraction-on-fsdsoundscapes","slug":"target-sound-extraction-on-fsdsoundscapes","dataset":"FSDSoundScapes","dataset_url":"/dataset/fsdsoundscapes","rows_in_archive":1,"metrics":["SI-SNRi"],"first_row_in_archive_order":{"model":"Waveformer","paper_title":"Real-Time Target Sound Extraction","paper_url":"/paper/real-time-target-sound-extraction","paper_date":"2022-11-04","arxiv_id":"2211.02250","code_links":[{"title":"vb000/waveformer","url":"https://github.com/vb000/waveformer"}],"syntology":{"n":7,"n_ran":3,"n_unverified":4,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/audioset","name":"AudioSet","full_name":"","num_papers_in_archive":744},{"url":"/dataset/audiocaps","name":"AudioCaps","full_name":"","num_papers_in_archive":279},{"url":"/dataset/fsdsoundscapes","name":"FSDSoundScapes","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/streaming-target-sound-extraction","name":"Streaming Target Sound Extraction"}],"parent_tasks":[{"url":"/task/audio-source-separation","name":"Audio Source Separation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":8,"of":8,"tagged_in_all":16,"items":[{"url":"/paper/dpm-tse-a-diffusion-probabilistic-model-for","title":"DPM-TSE: A Diffusion Probabilistic Model for Target Sound Extraction","date":"2023-10-06","arxiv_id":"2310.04567","repositories_listed":2,"syntology":null},{"url":"/paper/soloaudio-target-sound-extraction-with","title":"SoloAudio: Target Sound Extraction with Language-oriented Audio Diffusion Transformer","date":"2024-09-12","arxiv_id":"2409.08425","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-inspired-selective-state","title":"Cross-attention Inspired Selective State Space Models for Target Sound Extraction","date":"2024-09-07","arxiv_id":"2409.04803","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}},{"url":"/paper/can-all-variations-within-the-unified-mask","title":"Can all variations within the unified mask-based beamformer framework achieve identical peak extraction performance?","date":"2024-07-22","arxiv_id":"2407.15310","repositories_listed":1,"syntology":null},{"url":"/paper/clapsep-leveraging-contrastive-pre-trained","title":"CLAPSep: Leveraging Contrastive Pre-trained Model for Multi-Modal Query-Conditioned Target Sound Extraction","date":"2024-02-27","arxiv_id":"2402.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/semantic-hearing-programming-acoustic-scenes","title":"Semantic Hearing: Programming Acoustic Scenes with Binaural Hearables","date":"2023-11-01","arxiv_id":"2311.00320","repositories_listed":1,"syntology":null},{"url":"/paper/target-sound-extraction-with-variable-cross","title":"Target Sound Extraction with Variable Cross-modality Clues","date":"2023-03-15","arxiv_id":"2303.08372","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-target-sound-extraction","title":"Real-Time Target Sound Extraction","date":"2022-11-04","arxiv_id":"2211.02250","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_unverified":4,"n_pointer_only":0}}],"syntology_records":3,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}