{"url":"/dataset/audioset","name":"AudioSet","full_name":null,"description_markdown":"Audioset is an audio event dataset, which consists of over 2M human-annotated 10-second video clips. These clips are collected from YouTube, therefore many of which are in poor-quality and contain multiple sound-sources. A hierarchical ontology of 632 event classes is employed to annotate these data, which means that the same sound could be annotated as different labels. For example, the sound of barking is annotated as Animal, Pets, and Dog. All the videos are split into Evaluation/Balanced-Train/Unbalanced-Train set.\r\n\r\nSource: [Curriculum Audiovisual Learning](https://arxiv.org/abs/2001.09414)","description_withheld":null,"homepage":"https://research.google.com/audioset/index.html","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"Audio Set: An ontology and human-labeled dataset for audio events","first_author":null,"url":"https://doi.org/10.1109/ICASSP.2017.7952261"},"license":{"name":"CC BY 4.0","url":"https://research.google.com/audioset/download.html"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Audio Classification","url":"/task/audio-classification","datasets_with_task":"/datasets/task/audio-classification"},{"name":"Audio Source Separation","url":"/task/audio-source-separation","datasets_with_task":"/datasets/task/audio-source-separation"},{"name":"Target Sound Extraction","url":"/task/target-sound-extraction","datasets_with_task":"/datasets/task/target-sound-extraction"},{"name":"Zero-shot Audio Classification","url":"/task/zero-shot-audio-classification","datasets_with_task":"/datasets/task/zero-shot-audio-classification"},{"name":"Multi-modal Classification","url":"/task/multi-modal-classification","datasets_with_task":"/datasets/task/multi-modal-classification"},{"name":"Audio Tagging","url":"/task/audio-tagging","datasets_with_task":"/datasets/task/audio-tagging"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["AudioSet"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/agkphysics/AudioSet","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":744,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/audio-classification-on-audioset","task":"Audio Classification","dataset_variant":"AudioSet","rows":51,"metrics":["Test mAP","AUC","d-prime"],"first_row_in_archive_order":{"model":"OmniVec2","paper":"/paper/omnivec2-a-novel-transformer-based-network","metrics":{"Test mAP":"0.558"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/audio-tagging-on-audioset","task":"Audio Tagging","dataset_variant":"AudioSet","rows":11,"metrics":["mean average precision"],"first_row_in_archive_order":{"model":"CAV-MAE (Audio-Visual)","paper":"/paper/contrastive-audio-visual-masked-autoencoder","metrics":{"mean average precision":"0.512"},"code_links":[{"title":"yuangongnd/cav-mae","url":"https://github.com/yuangongnd/cav-mae"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/audio-source-separation-on-audioset","task":"Audio Source Separation","dataset_variant":"AudioSet","rows":2,"metrics":["SDR","SAR","SIR"],"first_row_in_archive_order":{"model":"ST-SED-SEP","paper":"/paper/zero-shot-audio-source-separation-through","metrics":{"SDR":"10.55"},"code_links":[{"title":"RetroCirce/Zero_Shot_Audio_Source_Separation","url":"https://github.com/RetroCirce/Zero_Shot_Audio_Source_Separation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-modal-classification-on-audioset","task":"Multi-modal Classification","dataset_variant":"AudioSet","rows":2,"metrics":["Average mAP"],"first_row_in_archive_order":{"model":"CAV-MAE","paper":"/paper/contrastive-audio-visual-masked-autoencoder","metrics":{"Average mAP":"0.512"},"code_links":[{"title":"yuangongnd/cav-mae","url":"https://github.com/yuangongnd/cav-mae"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/target-sound-extraction-on-audioset","task":"Target Sound Extraction","dataset_variant":"AudioSet","rows":1,"metrics":["SDRi","SI-SDRi"],"first_row_in_archive_order":{"model":"CLAPSep","paper":"/paper/clapsep-leveraging-contrastive-pre-trained","metrics":{"SDRi":"9.29","SI-SDRi":"8.44"},"code_links":[{"title":"aisaka0v0/clapsep","url":"https://github.com/aisaka0v0/clapsep"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/sslam-enhancing-self-supervised-models-with-1","title":"SSLAM: Enhancing Self-Supervised Models with Audio Mixtures for Polyphonic Soundscapes","date":"2025-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/m2d2-exploring-general-purpose-audio-language","title":"M2D2: Exploring General-purpose Audio-Language Representations Beyond CLAP","date":"2025-03-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/dass-distilled-audio-state-space-models-are","title":"DASS: Distilled Audio State Space Models Are Stronger and More Duration-Scalable Learners","date":"2024-07-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/m2d-clap-masked-modeling-duo-meets-clap-for","title":"M2D-CLAP: Masked Modeling Duo Meets CLAP for Learning General-purpose Audio-Language Representation","date":"2024-06-04","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/max-ast-combining-convolution-local-and","title":"MAX-AST: COMBINING CONVOLUTION, LOCAL AND GLOBAL SELF-ATTENTIONS FOR AUDIO EVENT CLASSIFICATION","date":"2024-04-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/masked-modeling-duo-towards-a-universal-audio","title":"Masked Modeling Duo: Towards a Universal Audio Pre-training Framework","date":"2024-04-09","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":7,"samples_unverified":0,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dtf-at-decoupled-time-frequency-audio","title":"DTF-AT: Decoupled Time-Frequency Audio Transformer for Event Classification","date":"2024-03-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/equiav-leveraging-equivariance-for-audio","title":"EquiAV: Leveraging Equivariance for Audio-Visual Contrastive Learning","date":"2024-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":6,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/clapsep-leveraging-contrastive-pre-trained","title":"CLAPSep: Leveraging Contrastive Pre-trained Model for Multi-Modal Query-Conditioned Target Sound Extraction","date":"2024-02-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/eat-self-supervised-pre-training-with","title":"EAT: Self-Supervised Pre-Training with Efficient Audio Transformer","date":"2024-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":13,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivec2-a-novel-transformer-based-network","title":"OmniVec2 - A Novel Transformer based Network for Large Scale Multimodal and Multitask Learning","date":"2024-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/omnivec-learning-robust-representations-with","title":"OmniVec: Learning robust representations with cross modal sharing","date":"2023-11-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dynamic-convolutional-neural-networks-as","title":"Dynamic Convolutional Neural Networks as Efficient Pre-trained Audio Models","date":"2023-10-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-audio-teacher-student","title":"Self-supervised Audio Teacher-Student Transformer for Both Clip-level and Frame-level Tasks","date":"2023-06-07","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/beats-audio-pre-training-with-acoustic","title":"BEATs: Audio Pre-Training with Acoustic Tokenizers","date":"2022-12-18","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":3,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/audiovisual-masked-autoencoders","title":"Audiovisual Masked Autoencoders","date":"2022-12-09","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/efficient-large-scale-audio-tagging-via","title":"Efficient Large-scale Audio Tagging via Transformer-to-CNN Knowledge Distillation","date":"2022-11-09","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/play-it-back-iterative-attention-for-audio","title":"Play It Back: Iterative Attention for Audio Recognition","date":"2022-10-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/contrastive-audio-visual-masked-autoencoder","title":"Contrastive Audio-Visual Masked Autoencoder","date":"2022-10-02","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uavm-a-unified-model-for-audio-visual","title":"UAVM: Towards Unifying Audio and Visual Models","date":"2022-07-29","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":6,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-audio-strikes-back-boosting","title":"End-to-End Audio Strikes Back: Boosting Augmentations Towards An Efficient Audio Classification Network","date":"2022-04-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hts-at-a-hierarchical-token-semantic-audio","title":"HTS-AT: A Hierarchical Token-Semantic Audio Transformer for Sound Classification and Detection","date":"2022-02-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-audio-source-separation-through","title":"Zero-shot Audio Source Separation through Query-based Learning from Weakly-labeled Data","date":"2021-12-15","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/conformer-based-self-supervised-learning-for","title":"Conformer-Based Self-Supervised Learning for Non-Speech Audio Tasks","date":"2021-10-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-training-of-audio-transformers-with","title":"Efficient Training of Audio Transformers with Patchout","date":"2021-10-11","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-bottlenecks-for-multimodal-fusion","title":"Attention Bottlenecks for Multimodal Fusion","date":"2021-06-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vatt-transformers-for-multimodal-self","title":"VATT: Transformers for Multimodal Self-Supervised Learning from Raw Video, Audio and Text","date":"2021-04-22","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ast-audio-spectrogram-transformer","title":"AST: Audio Spectrogram Transformer","date":"2021-04-05","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-format-contrastive-learning-of-audio","title":"Multi-Format Contrastive Learning of Audio Representations","date":"2021-03-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/perceiver-general-perception-with-iterative","title":"Perceiver: General Perception with Iterative Attention","date":"2021-03-04","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":55,"samples_ran":41,"samples_unverified":14,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/psla-improving-audio-event-classification","title":"PSLA: Improving Audio Tagging with Pretraining, Sampling, Labeling, and Aggregation","date":"2021-02-02","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/a-sequential-self-teaching-approach-for","title":"A Sequential Self Teaching Approach for Improving Generalization in Sound Event Recognition","date":"2020-06-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-multimodal-versatile-networks","title":"Self-Supervised MultiModal Versatile Networks","date":"2020-06-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/large-scale-audiovisual-learning-of-sounds","title":"Large Scale Audiovisual Learning of Sounds with Weakly Labeled Data","date":"2020-05-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/co-separating-sounds-of-visual-objects","title":"Co-Separating Sounds of Visual Objects","date":"2019-04-16","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/unsupervised-learning-of-semantic-audio","title":"Unsupervised Learning of Semantic Audio Representations","date":"2017-11-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/look-listen-and-learn","title":"Look, Listen and Learn","date":"2017-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/panns-large-scale-pretrained-audio-neural-1","title":"PANNs: Large-Scale Pretrained Audio Neural Networks for Audio Pattern Recognition","date":null,"rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":2,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":17,"samples_harvested":188,"samples_ran":104,"samples_unverified":84,"pointer_only_for_licence":27,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}