{"url":"/dataset/esc-50","name":"ESC-50","full_name":"ESC-50","description_markdown":"The **ESC-50** dataset is a labeled collection of 2000 environmental audio recordings suitable for benchmarking methods of environmental sound classification. It comprises 2000 5s-clips of 50 different classes across natural, human and domestic sounds, again, drawn from Freesound.org.\r\n\r\nSource: [The NIGENS General Sound Events Database](https://arxiv.org/abs/1902.08314)\r\nImage Source: [https://github.com/karolpiczak/ESC-50](https://github.com/karolpiczak/ESC-50)","description_withheld":null,"homepage":"https://github.com/karolpiczak/ESC-50","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"ESC: Dataset for Environmental Sound Classification","first_author":null,"url":"https://doi.org/10.1145/2733373.2806390"},"license":{"name":"CC BY-NC 3.0","url":"https://creativecommons.org/licenses/by-nc/3.0/"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Image Classification","url":"/task/image-classification","datasets_with_task":"/datasets/task/image-classification"},{"name":"Audio Classification","url":"/task/audio-classification","datasets_with_task":"/datasets/task/audio-classification"},{"name":"Few-Shot Audio Classification","url":"/task/few-shot-audio-classification","datasets_with_task":"/datasets/task/few-shot-audio-classification"},{"name":"Data Augmentation","url":"/task/data-augmentation","datasets_with_task":"/datasets/task/data-augmentation"},{"name":"Zero-shot Audio Classification","url":"/task/zero-shot-audio-classification","datasets_with_task":"/datasets/task/zero-shot-audio-classification"},{"name":"Environmental Sound Classification","url":"/task/environmental-sound-classification","datasets_with_task":"/datasets/task/environmental-sound-classification"},{"name":"Self-Supervised Audio Classification","url":"/task/self-supervised-audio-classification","datasets_with_task":"/datasets/task/self-supervised-audio-classification"},{"name":"Environment Sound Classification","url":"/task/environment-sound-classification","datasets_with_task":"/datasets/task/environment-sound-classification"},{"name":"Zero-Shot Environment Sound Classification","url":"/task/zero-shot-environment-sound-classification","datasets_with_task":"/datasets/task/zero-shot-environment-sound-classification"}],"languages":[],"variants":["ESC-50"],"data_loaders":[{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/esc-50-dataset","frameworks":["tf","pytorch"]},{"repo":"https://github.com/karolpiczak/ESC-50","url":"https://github.com/karolpiczak/ESC-50","frameworks":[]}],"num_papers_in_archive":387,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/audio-classification-on-esc-50","task":"Audio Classification","dataset_variant":"ESC-50","rows":29,"metrics":["Top-1 Accuracy","PRE-TRAINING DATASET","Accuracy (5-fold)"],"first_row_in_archive_order":{"model":"OmniVec2","paper":"/paper/omnivec2-a-novel-transformer-based-network","metrics":{"Accuracy (5-fold)":"99.1","PRE-TRAINING DATASET":"Multiple","Top-1 Accuracy":"99.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-audio-classification-on-esc-50","task":"Few-Shot Audio Classification","dataset_variant":"ESC-50","rows":10,"metrics":["Top-1 Accuracy(5-Way-1-Shot)"],"first_row_in_archive_order":{"model":"Meta-Curvature (CRNN)","paper":"/paper/metaaudio-a-few-shot-audio-classification","metrics":{"Top-1 Accuracy(5-Way-1-Shot)":"76.17 +- 0.41"},"code_links":[{"title":"cheggan/metaaudio-a-few-shot-audio-classification-benchmark","url":"https://github.com/cheggan/metaaudio-a-few-shot-audio-classification-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/environmental-sound-classification-on-esc-50","task":"Environmental Sound Classification","dataset_variant":"ESC-50","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"AudioCLIP","paper":"/paper/audioclip-extending-clip-to-image-text-and","metrics":{"Accuracy":"97.15"},"code_links":[{"title":"iver56/audiomentations","url":"https://github.com/iver56/audiomentations"},{"title":"asteroid-team/torch-audiomentations","url":"https://github.com/asteroid-team/torch-audiomentations"},{"title":"AndreyGuzhov/AudioCLIP","url":"https://github.com/AndreyGuzhov/AudioCLIP"},{"title":"julirao/whisper_audio_classification","url":"https://github.com/julirao/whisper_audio_classification"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-classification-on-esc-50","task":"Image Classification","dataset_variant":"ESC-50","rows":1,"metrics":["Top 1 Accuracy"],"first_row_in_archive_order":{"model":"SDGM-D","paper":"/paper/performance-of-gaussian-mixture-model","metrics":{"Top 1 Accuracy":"87"},"code_links":[{"title":"cvmlmu/dgmmc","url":"https://github.com/cvmlmu/dgmmc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/m2d2-exploring-general-purpose-audio-language","title":"M2D2: Exploring General-purpose Audio-Language Representations Beyond CLAP","date":"2025-03-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/masked-latent-prediction-and-classification","title":"Masked Latent Prediction and Classification for Self-Supervised Audio Representation Learning","date":"2025-02-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lhgnn-local-higher-order-graph-neural","title":"LHGNN: Local-Higher Order Graph Neural Networks For Audio Classification and Tagging","date":"2025-01-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/performance-of-gaussian-mixture-model","title":"Performance of Gaussian Mixture Model Classifiers on Embedded Feature Spaces","date":"2024-10-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/m2d-clap-masked-modeling-duo-meets-clap-for","title":"M2D-CLAP: Masked Modeling Duo Meets CLAP for Learning General-purpose Audio-Language Representation","date":"2024-06-04","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/masked-modeling-duo-towards-a-universal-audio","title":"Masked Modeling Duo: Towards a Universal Audio Pre-training Framework","date":"2024-04-09","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":7,"samples_unverified":0,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/eat-self-supervised-pre-training-with","title":"EAT: Self-Supervised Pre-Training with Efficient Audio Transformer","date":"2024-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":13,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivec2-a-novel-transformer-based-network","title":"OmniVec2 - A Novel Transformer based Network for Large Scale Multimodal and Multitask Learning","date":"2024-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/omnivec-learning-robust-representations-with","title":"OmniVec: Learning robust representations with cross modal sharing","date":"2023-11-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dynamic-convolutional-neural-networks-as","title":"Dynamic Convolutional Neural Networks as Efficient Pre-trained Audio Models","date":"2023-10-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mt-slvr-multi-task-self-supervised-learning","title":"MT-SLVR: Multi-Task Self-Supervised Learning for Transformation In(Variant) Representations","date":"2023-05-29","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/beats-audio-pre-training-with-acoustic","title":"BEATs: Audio Pre-Training with Acoustic Tokenizers","date":"2022-12-18","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":3,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-large-scale-audio-tagging-via","title":"Efficient Large-scale Audio Tagging via Transformer-to-CNN Knowledge Distillation","date":"2022-11-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lerac-learning-rate-curriculum","title":"Learning Rate Curriculum","date":"2022-05-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-audio-strikes-back-boosting","title":"End-to-End Audio Strikes Back: Boosting Augmentations Towards An Efficient Audio Classification Network","date":"2022-04-25","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/metaaudio-a-few-shot-audio-classification","title":"MetaAudio: A Few-Shot Audio Classification Benchmark","date":"2022-04-05","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/septr-separable-transformer-for-audio","title":"SepTr: Separable Transformer for Audio Spectrogram Processing","date":"2022-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hts-at-a-hierarchical-token-semantic-audio","title":"HTS-AT: A Hierarchical Token-Semantic Audio Transformer for Sound Classification and Detection","date":"2022-02-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/audioclip-extending-clip-to-image-text-and","title":"AudioCLIP: Extending CLIP to Image, Text and Audio","date":"2021-06-24","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ast-audio-spectrogram-transformer","title":"AST: Audio Spectrogram Transformer","date":"2021-04-05","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-format-contrastive-learning-of-audio","title":"Multi-Format Contrastive Learning of Audio Representations","date":"2021-03-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/environmental-sound-classification-on-the","title":"Environmental Sound Classification on the Edge: A Pipeline for Deep Acoustic Networks on Extremely Resource-Constrained Devices","date":"2021-03-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/audio-visual-instance-discrimination-with","title":"Audio-Visual Instance Discrimination with Cross-Modal Agreement","date":"2020-04-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-learning-by-cross-modal-audio","title":"Self-Supervised Learning by Cross-Modal Audio-Video Clustering","date":"2019-11-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cooperative-learning-of-audio-and-video","title":"Cooperative Learning of Audio and Video Models from Self-Supervised Synchronization","date":"2018-06-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/look-listen-and-learn","title":"Look, Listen and Learn","date":"2017-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":64,"samples_ran":29,"samples_unverified":35,"pointer_only_for_licence":8,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}