{"url":"/dataset/fsd50k","name":"FSD50K","full_name":"Freesound Database 50K","description_markdown":"Freesound Dataset 50k (or **FSD50K** for short) is an open dataset of human-labeled sound events containing 51,197 Freesound clips unequally distributed in 200 classes drawn from the AudioSet Ontology. FSD50K has been created at the Music Technology Group of Universitat Pompeu Fabra. It consists mainly of sound events produced by physical sound sources and production mechanisms, including human sounds, sounds of things, animals, natural sounds, musical instruments and more.\n\nSource: [https://zenodo.org/record/4060432](https://zenodo.org/record/4060432)\nImage Source: [https://labs.freesound.org/datasets/](https://labs.freesound.org/datasets/)","description_withheld":null,"homepage":"https://zenodo.org/record/4060432","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/fsd50k-an-open-dataset-of-human-labeled-sound","title":"FSD50K: An Open Dataset of Human-Labeled Sound Events","first_author":"Eduardo Fonseca","url":null},"license":{"name":"Other (Attribution)","url":"https://zenodo.org/record/4060432"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Domain Adaptation","url":"/task/domain-adaptation","datasets_with_task":"/datasets/task/domain-adaptation"},{"name":"Audio Classification","url":"/task/audio-classification","datasets_with_task":"/datasets/task/audio-classification"},{"name":"Contrastive Learning","url":"/task/contrastive-learning","datasets_with_task":"/datasets/task/contrastive-learning"},{"name":"Environmental Sound Classification","url":"/task/environmental-sound-classification","datasets_with_task":"/datasets/task/environmental-sound-classification"}],"languages":[],"variants":["FSD50K"],"data_loaders":[],"num_papers_in_archive":155,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/audio-classification-on-fsd50k","task":"Audio Classification","dataset_variant":"FSD50K","rows":10,"metrics":["mAP","Mean AP"],"first_row_in_archive_order":{"model":"ONE-PEACE","paper":"/paper/one-peace-exploring-one-general","metrics":{"mAP":"69.7"},"code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"OFA-Sys/ONE-PEACE","url":"https://github.com/OFA-Sys/ONE-PEACE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/environmental-sound-classification-on-fsd50k","task":"Environmental Sound Classification","dataset_variant":"FSD50K","rows":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"[ABT] AudioNTT","paper":"/paper/audio-barlow-twins-self-supervised-audio","metrics":{"mAP":"0.474"},"code_links":[{"title":"jonahanton/ssl_audio","url":"https://github.com/jonahanton/ssl_audio"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/masked-latent-prediction-and-classification","title":"Masked Latent Prediction and Classification for Self-Supervised Audio Representation Learning","date":"2025-02-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lhgnn-local-higher-order-graph-neural","title":"LHGNN: Local-Higher Order Graph Neural Networks For Audio Classification and Tagging","date":"2025-01-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dynamic-convolutional-neural-networks-as","title":"Dynamic Convolutional Neural Networks as Efficient Pre-trained Audio Models","date":"2023-10-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/audio-barlow-twins-self-supervised-audio","title":"Audio Barlow Twins: Self-Supervised Audio Representation Learning","date":"2022-09-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-knowledge-distillation-for-on-device","title":"Temporal Knowledge Distillation for On-device Audio Classification","date":"2021-10-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-training-of-audio-transformers-with","title":"Efficient Training of Audio Transformers with Patchout","date":"2021-10-11","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/audio-transformers-transformer-architectures","title":"Audio Transformers","date":"2021-05-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/psla-improving-audio-event-classification","title":"PSLA: Improving Audio Tagging with Pretraining, Sampling, Labeling, and Aggregation","date":"2021-02-02","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}