{"url":"/dataset/fsdsoundscapes","name":"FSDSoundScapes","full_name":null,"description_markdown":"A synthetic sound mixture specification dataset for the Target Sound Extraction (TSE) task.  Dataset samples consist of a [.jams](https://jams.readthedocs.io/en/stable/) file specifying the mixture components, and a metadata file with target labels. Mixtures are 6 seconds long and contain 3-5 unique foreground sounds over a 6 second long background sound. Each sample is provided with 3 target labels, and sounds corresponding to all target labels are guaranteed to be present in the mixture. [FSDKaggle2018](https://zenodo.org/record/2552860) is used as the source for foreground sounds and [TAU Urban Acoustic Scenes 2019](https://dcase.community/challenge2019/task-acoustic-scene-classification) is used as the source for background sounds.\r\n\r\n##### Split\r\nTrain: 50K  \r\nVal: 5K  \r\nTest: 10K","description_withheld":null,"homepage":"https://github.com/vb000/Waveformer#dataset","introduced_date":"2022-11-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/real-time-target-sound-extraction","title":"Real-Time Target Sound Extraction","first_author":"Bandhav Veluri","url":null},"license":{"name":"MIT License","url":"https://github.com/vb000/Waveformer/blob/main/LICENSE"},"modalities":[],"tasks":[{"name":"Target Sound Extraction","url":"/task/target-sound-extraction","datasets_with_task":"/datasets/task/target-sound-extraction"},{"name":"Streaming Target Sound Extraction","url":"/task/streaming-target-sound-extraction","datasets_with_task":"/datasets/task/streaming-target-sound-extraction"}],"languages":[],"variants":["FSDSoundScapes"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/streaming-target-sound-extraction-on","task":"Streaming Target Sound Extraction","dataset_variant":"FSDSoundScapes","rows":1,"metrics":["SI-SNRi"],"first_row_in_archive_order":{"model":"Waveformer","paper":"/paper/real-time-target-sound-extraction","metrics":{"SI-SNRi":"9.43"},"code_links":[{"title":"vb000/waveformer","url":"https://github.com/vb000/waveformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/target-sound-extraction-on-fsdsoundscapes","task":"Target Sound Extraction","dataset_variant":"FSDSoundScapes","rows":1,"metrics":["SI-SNRi"],"first_row_in_archive_order":{"model":"Waveformer","paper":"/paper/real-time-target-sound-extraction","metrics":{"SI-SNRi":"9.43"},"code_links":[{"title":"vb000/waveformer","url":"https://github.com/vb000/waveformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/real-time-target-sound-extraction","title":"Real-Time Target Sound Extraction","date":"2022-11-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}