{"url":"/task/sound-event-detection","name":"Sound Event Detection","slug":"sound-event-detection","description_markdown":"**Sound Event Detection** (SED) is the task of recognizing the sound events and their respective temporal start and end time in a recording. Sound events in real life do not always occur in isolation, but tend to considerably overlap with each other. Recognizing such overlapping sound events is referred as polyphonic SED.\n\n\n<span class=\"description-source\">Source: [A report on sound event detection with different binaural features ](https://arxiv.org/abs/1710.02997)</span>","categories":[{"name":"Audio","url":"/area/audio"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":194,"papers_with_code":92,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":20,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/sound-event-detection-on-desed","slug":"sound-event-detection-on-desed","dataset":"DESED","dataset_url":"/dataset/desed","rows_in_archive":13,"metrics":["event-based F1 score","PSDS1","PSDS2"],"first_row_in_archive_order":{"model":"ATST-SED","paper_title":"Fine-tune the pretrained ATST model for sound event detection","paper_url":"/paper/fine-tune-the-pretrained-atst-model-for-sound","paper_date":"2023-09-15","arxiv_id":"2309.08153","code_links":[{"title":"Audio-WestlakeU/ATST-SED","url":"https://github.com/Audio-WestlakeU/ATST-SED"}],"syntology":{"n":10,"n_ran":10,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/sound-event-detection-on-l3das21","slug":"sound-event-detection-on-l3das21","dataset":"L3DAS21","dataset_url":"/dataset/l3das21","rows_in_archive":5,"metrics":["Error Rate","SED-score","F-Score"],"first_row_in_archive_order":{"model":"PHC SEDnet n=2","paper_title":"PHNNs: Lightweight Neural Networks via Parameterized Hypercomplex Convolutions","paper_url":"/paper/lightweight-convolutional-neural-networks-by","paper_date":"2021-10-08","arxiv_id":"2110.04176","code_links":[{"title":"elegan23/hypernets","url":"https://github.com/elegan23/hypernets"},{"title":"eleGAN23/QVAE","url":"https://github.com/eleGAN23/QVAE"},{"title":"ispamm/hi2i","url":"https://github.com/ispamm/hi2i"},{"title":"eleGAN23/QGAN","url":"https://github.com/eleGAN23/QGAN"}],"syntology":{"n":34,"n_ran":5,"n_unverified":29,"n_pointer_only":0}}},{"leaderboard":"/sota/sound-event-detection-on-wilddesed","slug":"sound-event-detection-on-wilddesed","dataset":"WildDESED","dataset_url":"/dataset/wilddesed","rows_in_archive":5,"metrics":["PSDS1 (-5dB)","PSDS1 (0dB)","PSDS1 (5dB)","PSDS1 (10dB)","PSDS1 (Clean)"],"first_row_in_archive_order":{"model":"CRNN (with BEATs + Separation)","paper_title":"Leveraging LLM and Text-Queried Separation for Noise-Robust Sound Event Detection","paper_url":"/paper/leveraging-llm-and-text-queried-separation","paper_date":"2024-11-02","arxiv_id":"2411.01174","code_links":[{"title":"apple-yinhan/noise-robust-sed","url":"https://github.com/apple-yinhan/noise-robust-sed"}],"syntology":null}},{"leaderboard":"/sota/sound-event-detection-on-mivia-audio-events","slug":"sound-event-detection-on-mivia-audio-events","dataset":"Mivia Audio Events","dataset_url":null,"rows_in_archive":1,"metrics":["Rank-1 Recognition Rate"],"first_row_in_archive_order":{"model":"DENet","paper_title":"DENet: a deep architecture for audio surveillance applications","paper_url":"/paper/denet-a-deep-architecture-for-audio","paper_date":"2021-01-11","arxiv_id":null,"code_links":[{"title":"MiviaLab/DENet","url":"https://github.com/MiviaLab/DENet"}],"syntology":null}},{"leaderboard":"/sota/sound-event-detection-on-mivia-road-events","slug":"sound-event-detection-on-mivia-road-events","dataset":"Mivia Road Events","dataset_url":null,"rows_in_archive":1,"metrics":["Rank-1 Recognition Rate"],"first_row_in_archive_order":{"model":"DENet","paper_title":"DENet: a deep architecture for audio surveillance applications","paper_url":"/paper/denet-a-deep-architecture-for-audio","paper_date":"2021-01-11","arxiv_id":null,"code_links":[{"title":"MiviaLab/DENet","url":"https://github.com/MiviaLab/DENet"}],"syntology":null}}],"datasets":[{"url":"/dataset/dcase-2016","name":"DCASE 2016","full_name":"DCASE 2016","num_papers_in_archive":31},{"url":"/dataset/fsdnoisy18k","name":"FSDnoisy18k","full_name":"FSDnoisy18k","num_papers_in_archive":19},{"url":"/dataset/desed","name":"DESED","full_name":"Domestic environment sound event detection","num_papers_in_archive":17},{"url":"/dataset/l3das22","name":"L3DAS22","full_name":"","num_papers_in_archive":13},{"url":"/dataset/dcase-2013","name":"DCASE 2013","full_name":"DCASE 2013","num_papers_in_archive":11},{"url":"/dataset/tut-sound-events-2017","name":"TUT Sound Events 2017","full_name":"TUT Sound Events 2017","num_papers_in_archive":8},{"url":"/dataset/tut-sed-synthetic-2016","name":"TUT-SED Synthetic 2016","full_name":"TUT-SED Synthetic 2016","num_papers_in_archive":7},{"url":"/dataset/l3das21","name":"L3DAS21","full_name":"","num_papers_in_archive":6},{"url":"/dataset/tau-nigens-spatial-sound-events-2021","name":"TAU-NIGENS Spatial Sound Events 2021","full_name":"TAU-NIGENS Spatial Sound Events 2021","num_papers_in_archive":4},{"url":"/dataset/birdvox-full-night","name":"BirdVox-full-night","full_name":"BirdVox-full-night","num_papers_in_archive":3},{"url":"/dataset/dcase-2017","name":"DCASE 2017","full_name":"DCASE 2017","num_papers_in_archive":2},{"url":"/dataset/singa-pura","name":"SINGA:PURA","full_name":"SINGApore: Polyphonic URban Audio","num_papers_in_archive":2},{"url":"/dataset/tau-nigens-spatial-sound-events-2020","name":"TAU-NIGENS Spatial Sound Events 2020","full_name":"TAU-NIGENS Spatial Sound Events 2020","num_papers_in_archive":2},{"url":"/dataset/urban-sed","name":"URBAN-SED","full_name":"URBAN-SED","num_papers_in_archive":2},{"url":"/dataset/voice","name":"VOICe","full_name":"","num_papers_in_archive":2},{"url":"/dataset/wilddesed","name":"WildDESED","full_name":"Wild Domestic Environment Sound Event Detection","num_papers_in_archive":2},{"url":"/dataset/biosed-acpd","name":"BIOSED-ACPD","full_name":"","num_papers_in_archive":1},{"url":"/dataset/erhupt","name":"ErhuPT","full_name":"Erhu Playing Technique Dataset","num_papers_in_archive":1},{"url":"/dataset/usm-sed","name":"USM-SED","full_name":"","num_papers_in_archive":1},{"url":"/dataset/tut-sound-events-2018","name":"TUT Sound Events 2018","full_name":"TUT Sound Events 2018","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":92,"tagged_in_all":194,"items":[{"url":"/paper/towards-deep-learning-models-resistant-to","title":"Towards Deep Learning Models Resistant to Adversarial Attacks","date":"2017-06-19","arxiv_id":"1706.06083","repositories_listed":59,"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":13}},{"url":"/paper/lightweight-convolutional-neural-networks-by","title":"PHNNs: Lightweight Neural Networks via Parameterized Hypercomplex Convolutions","date":"2021-10-08","arxiv_id":"2110.04176","repositories_listed":4,"syntology":{"n":34,"n_ran":5,"n_unverified":29,"n_pointer_only":0}},{"url":"/paper/wavcaps-a-chatgpt-assisted-weakly-labelled","title":"WavCaps: A ChatGPT-Assisted Weakly-Labelled Audio Captioning Dataset for Audio-Language Multimodal Research","date":"2023-03-30","arxiv_id":"2303.17395","repositories_listed":3,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/effective-pre-training-of-audio-transformers","title":"Effective Pre-Training of Audio Transformers for Sound Event Detection","date":"2024-09-14","arxiv_id":"2409.09546","repositories_listed":2,"syntology":null},{"url":"/paper/mind-the-domain-gap-a-systematic-analysis-on","title":"Mind the Domain Gap: a Systematic Analysis on Bioacoustic Sound Event Detection","date":"2024-03-27","arxiv_id":"2403.18638","repositories_listed":2,"syntology":null},{"url":"/paper/performance-and-energy-balance-a","title":"Performance and energy balance: a comprehensive study of state-of-the-art sound event detection systems","date":"2023-10-05","arxiv_id":"2310.03455","repositories_listed":2,"syntology":null},{"url":"/paper/self-supervised-audio-teacher-student","title":"Self-supervised Audio Teacher-Student Transformer for Both Clip-level and Frame-level Tasks","date":"2023-06-07","arxiv_id":"2306.04186","repositories_listed":2,"syntology":null},{"url":"/paper/sound-event-localization-and-detection-for","title":"Sound Event Localization and Detection for Real Spatial Sound Scenes: Event-Independent Network and Data Augmentation Chains","date":"2022-09-05","arxiv_id":"2209.01802","repositories_listed":2,"syntology":null},{"url":"/paper/rct-random-consistency-training-for-semi","title":"RCT: Random Consistency Training for Semi-supervised Sound Event Detection","date":"2021-10-21","arxiv_id":"2110.11144","repositories_listed":2,"syntology":null},{"url":"/paper/couple-learning-mean-teacher-method-with","title":"Couple Learning for semi-supervised sound event detection","date":"2021-10-12","arxiv_id":"2110.05809","repositories_listed":2,"syntology":null},{"url":"/paper/accdoa-activity-coupled-cartesian-direction","title":"ACCDOA: Activity-Coupled Cartesian Direction of Arrival Representation for Sound Event Localization and Detection","date":"2020-10-29","arxiv_id":"2010.15306","repositories_listed":2,"syntology":null},{"url":"/paper/seld-tcn-sound-event-localization-detection","title":"SELD-TCN: Sound Event Localization & Detection via Temporal Convolutional Networks","date":"2020-03-03","arxiv_id":"2003.01609","repositories_listed":2,"syntology":null},{"url":"/paper/learning-sound-event-classifiers-from-web","title":"Learning Sound Event Classifiers from Web Audio with Noisy Labels","date":"2019-01-04","arxiv_id":"1901.01189","repositories_listed":2,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/adaptive-pooling-operators-for-weakly-labeled","title":"Adaptive pooling operators for weakly labeled sound event detection","date":"2018-04-26","arxiv_id":"1804.10070","repositories_listed":2,"syntology":null},{"url":"/paper/recurrent-neural-networks-for-polyphonic","title":"Recurrent Neural Networks for Polyphonic Sound Event Detection in Real Life Recordings","date":"2016-04-04","arxiv_id":"1604.00861","repositories_listed":2,"syntology":null},{"url":"/paper/hybrid-disagreement-diversity-active-learning","title":"Hybrid Disagreement-Diversity Active Learning for Bioacoustic Sound Event Detection","date":"2025-05-27","arxiv_id":"2505.20956","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-attention-pooling-for-frequency","title":"Temporal Attention Pooling for Frequency Dynamic Convolution in Sound Event Detection","date":"2025-04-17","arxiv_id":"2504.12670","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-performance-complexity-trade-offs","title":"Exploring Performance-Complexity Trade-Offs in Sound Event Detection Models","date":"2025-03-14","arxiv_id":"2503.11373","repositories_listed":1,"syntology":null},{"url":"/paper/jitter-jigsaw-temporal-transformer-for-event","title":"JiTTER: Jigsaw Temporal Transformer for Event Reconstruction for Self-Supervised Sound Event Detection","date":"2025-02-28","arxiv_id":"2502.20857","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-llm-and-text-queried-separation","title":"Leveraging LLM and Text-Queried Separation for Noise-Robust Sound Event Detection","date":"2024-11-02","arxiv_id":"2411.01174","repositories_listed":1,"syntology":null},{"url":"/paper/prototype-based-masked-audio-model-for-self","title":"Prototype based Masked Audio Model for Self-Supervised Learning of Sound Event Detection","date":"2024-09-26","arxiv_id":"2409.17656","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-text-queried-sound-event-detection","title":"Exploring Text-Queried Sound Event Detection with Audio Source Separation","date":"2024-09-20","arxiv_id":"2409.13292","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/the-sounds-of-home-a-speech-removed","title":"The Sounds of Home: A Speech-Removed Residential Audio Dataset for Sound Event Detection","date":"2024-09-17","arxiv_id":"2409.11262","repositories_listed":1,"syntology":null},{"url":"/paper/mtda-hsed-mutual-assistance-tuning-and-dual","title":"MTDA-HSED: Mutual-Assistance Tuning and Dual-Branch Aggregating for Heterogeneous Sound Event Detection","date":"2024-09-10","arxiv_id":"2409.06196","repositories_listed":1,"syntology":null},{"url":"/paper/mat-sed-amasked-audio-transformer-with-masked","title":"MAT-SED: A Masked Audio Transformer with Masked-Reconstruction Based Pre-training for Sound Event Detection","date":"2024-08-16","arxiv_id":"2408.08673","repositories_listed":1,"syntology":null},{"url":"/paper/multi-iteration-multi-stage-fine-tuning-of","title":"Multi-Iteration Multi-Stage Fine-Tuning of Transformers for Sound Event Detection with Heterogeneous Datasets","date":"2024-07-17","arxiv_id":"2407.12997","repositories_listed":1,"syntology":null},{"url":"/paper/improving-audio-spectrogram-transformers-for","title":"Improving Audio Spectrogram Transformers for Sound Event Detection Through Multi-Stage Training","date":"2024-07-17","arxiv_id":"2408.00791","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/wilddesed-an-llm-powered-dataset-for-wild","title":"WildDESED: An LLM-Powered Dataset for Wild Domestic Environment Sound Event Detection System","date":"2024-07-04","arxiv_id":"2407.03656","repositories_listed":1,"syntology":null},{"url":"/paper/self-training-and-ensembling-frequency","title":"Self Training and Ensembling Frequency Dependent Networks with Coarse Prediction Pooling and Sound Event Bounding Boxes","date":"2024-06-22","arxiv_id":"2406.15725","repositories_listed":1,"syntology":null},{"url":"/paper/pushing-the-limit-of-sound-event-detection","title":"Pushing the Limit of Sound Event Detection with Multi-Dilated Frequency Dynamic Convolution","date":"2024-06-19","arxiv_id":"2406.13312","repositories_listed":1,"syntology":null}],"syntology_records":6,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}