{"url":"/task/sound-event-localization-and-detection","name":"Sound Event Localization and Detection","slug":"sound-event-localization-and-detection","description_markdown":"Given multichannel audio input, a sound event detection and localization (SELD) system outputs a temporal activation track for each of the target sound classes, along with one or more corresponding spatial trajectories when the track indicates activity. This results in a spatio-temporal characterization of the acoustic scene that can be used in a wide range of machine cognition tasks, such as inference on the type of environment, self-localization, navigation without visual input or with occluded targets, tracking of specific types of sound sources, smart-home applications, scene visualization systems, and audio surveillance, among others.","categories":[{"name":"Audio","url":"/area/audio"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":65,"papers_with_code":35,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/sound-event-localization-and-detection-on-1","slug":"sound-event-localization-and-detection-on-1","dataset":"STARSS22","dataset_url":"/dataset/starss22","rows_in_archive":2,"metrics":["Class-dependent localization error","Class-dependent localization recall","location-dependent F1-score (macro)","location-dependent F1-score (micro)","Localization-dependent error rate (20°)"],"first_row_in_archive_order":{"model":"Baseline (FOA)","paper_title":"STARSS22: A dataset of spatial recordings of real scenes with spatiotemporal annotations of sound events","paper_url":"/paper/starss22-a-dataset-of-spatial-recordings-of","paper_date":"2022-06-04","arxiv_id":"2206.01948","code_links":[{"title":"sharathadavanne/seld-dcase2022","url":"https://github.com/sharathadavanne/seld-dcase2022"},{"title":"prerak23/dir_srcmic_doa","url":"https://github.com/prerak23/dir_srcmic_doa"}],"syntology":null}},{"leaderboard":"/sota/sound-event-localization-and-detection-on-2","slug":"sound-event-localization-and-detection-on-2","dataset":"PodcastFillers","dataset_url":"/dataset/podcastfillers","rows_in_archive":2,"metrics":["event-based F1 score"],"first_row_in_archive_order":{"model":"AVC-FillerNet","paper_title":"Filler Word Detection and Classification: A Dataset and Benchmark","paper_url":"/paper/filler-word-detection-and-classification-a","paper_date":"2022-03-28","arxiv_id":"2203.15135","code_links":[{"title":"gzhu06/PodcastFillers_Utils","url":"https://github.com/gzhu06/PodcastFillers_Utils"}],"syntology":null}},{"leaderboard":"/sota/sound-event-localization-and-detection-on","slug":"sound-event-localization-and-detection-on","dataset":"L3DAS21","dataset_url":"/dataset/l3das21","rows_in_archive":1,"metrics":["SELD score"],"first_row_in_archive_order":{"model":"DualQSELD-TCN (parallel)","paper_title":"Dual Quaternion Ambisonics Array for Six-Degree-of-Freedom Acoustic Representation","paper_url":"/paper/dual-quaternion-ambisonics-array-for-six","paper_date":"2022-04-04","arxiv_id":"2204.01851","code_links":[{"title":"ispamm/dualqseld-tcn","url":"https://github.com/ispamm/dualqseld-tcn"}],"syntology":null}},{"leaderboard":"/sota/sound-event-localization-and-detection-on-3","slug":"sound-event-localization-and-detection-on-3","dataset":"RWCP Sound Scene Database","dataset_url":"/dataset/rwcp-sound-scene-database","rows_in_archive":1,"metrics":["accuracy"],"first_row_in_archive_order":{"model":"STL-SNN","paper_title":"A Synapse-Threshold Synergistic Learning Approach for Spiking Neural Networks","paper_url":"/paper/a-synapse-threshold-synergistic-learning","paper_date":"2022-06-10","arxiv_id":"2206.06129","code_links":[{"title":"sunhongze/STL-SNN","url":"https://github.com/sunhongze/STL-SNN"}],"syntology":null}},{"leaderboard":"/sota/sound-event-localization-and-detection-on-tau-1","slug":"sound-event-localization-and-detection-on-tau-1","dataset":"TAU-NIGENS Spatial Sound Events 2021","dataset_url":"/dataset/tau-nigens-spatial-sound-events-2021","rows_in_archive":1,"metrics":["ER≤20°","F1≤20°","LE-CD","LR-CD"],"first_row_in_archive_order":{"model":"SALSA-FOA","paper_title":"SALSA: Spatial Cue-Augmented Log-Spectrogram Features for Polyphonic Sound Event Localization and Detection","paper_url":"/paper/salsa-spatial-cue-augmented-log-spectrogram","paper_date":"2021-10-01","arxiv_id":"2110.00275","code_links":[{"title":"thomeou/SALSA","url":"https://github.com/thomeou/SALSA"}],"syntology":null}}],"datasets":[{"url":"/dataset/starss23","name":"STARSS23","full_name":"STARSS23: An Audio-Visual Dataset of Spatial Recordings of Real Scenes with Spatiotemporal Annotations of Sound Events","num_papers_in_archive":21},{"url":"/dataset/starss22","name":"STARSS22","full_name":"Sony-TAu Realistic Spatial Soundscapes 2022","num_papers_in_archive":19},{"url":"/dataset/l3das22","name":"L3DAS22","full_name":"","num_papers_in_archive":13},{"url":"/dataset/l3das21","name":"L3DAS21","full_name":"","num_papers_in_archive":6},{"url":"/dataset/podcastfillers","name":"PodcastFillers","full_name":"","num_papers_in_archive":5},{"url":"/dataset/tau-nigens-spatial-sound-events-2021","name":"TAU-NIGENS Spatial Sound Events 2021","full_name":"TAU-NIGENS Spatial Sound Events 2021","num_papers_in_archive":4},{"url":"/dataset/tau-nigens-spatial-sound-events-2020","name":"TAU-NIGENS Spatial Sound Events 2020","full_name":"TAU-NIGENS Spatial Sound Events 2020","num_papers_in_archive":2},{"url":"/dataset/bgg-dataset","name":"BGG dataset","full_name":"PUBG Gun Sound Dataset","num_papers_in_archive":1},{"url":"/dataset/rwcp-sound-scene-database","name":"RWCP Sound Scene Database","full_name":"RWCP Sound Scene Database","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":35,"tagged_in_all":65,"items":[{"url":"/paper/salsa-lite-a-fast-and-effective-feature-for","title":"SALSA-Lite: A Fast and Effective Feature for Polyphonic Sound Event Localization and Detection with Microphone Arrays","date":"2021-11-16","arxiv_id":"2111.08192","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/overview-and-evaluation-of-sound-event","title":"Overview and Evaluation of Sound Event Localization and Detection in DCASE 2019","date":"2020-09-06","arxiv_id":"2009.02792","repositories_listed":4,"syntology":null},{"url":"/paper/pseldnets-pre-trained-neural-networks-on","title":"PSELDNets: Pre-trained Neural Networks on Large-scale Synthetic Datasets for Sound Event Localization and Detection","date":"2024-11-10","arxiv_id":"2411.06399","repositories_listed":2,"syntology":null},{"url":"/paper/sound-event-localization-and-detection-for","title":"Sound Event Localization and Detection for Real Spatial Sound Scenes: Event-Independent Network and Data Augmentation Chains","date":"2022-09-05","arxiv_id":"2209.01802","repositories_listed":2,"syntology":null},{"url":"/paper/starss22-a-dataset-of-spatial-recordings-of","title":"STARSS22: A dataset of spatial recordings of real scenes with spatiotemporal annotations of sound events","date":"2022-06-04","arxiv_id":"2206.01948","repositories_listed":2,"syntology":null},{"url":"/paper/multi-accdoa-localizing-and-detecting","title":"Multi-ACCDOA: Localizing and Detecting Overlapping Sounds from the Same Class with Auxiliary Duplicating Permutation Invariant Training","date":"2021-10-14","arxiv_id":"2110.07124","repositories_listed":2,"syntology":null},{"url":"/paper/accdoa-activity-coupled-cartesian-direction","title":"ACCDOA: Activity-Coupled Cartesian Direction of Arrival Representation for Sound Event Localization and Detection","date":"2020-10-29","arxiv_id":"2010.15306","repositories_listed":2,"syntology":null},{"url":"/paper/a-dataset-of-reverberant-spatial-sound-scenes","title":"A Dataset of Reverberant Spatial Sound Scenes with Moving Sources for Sound Event Localization and Detection","date":"2020-06-02","arxiv_id":"2006.01919","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/seld-tcn-sound-event-localization-detection","title":"SELD-TCN: Sound Event Localization & Detection via Temporal Convolutional Networks","date":"2020-03-03","arxiv_id":"2003.01609","repositories_listed":2,"syntology":null},{"url":"/paper/reverberation-based-features-for-sound-event","title":"Reverberation-based Features for Sound Event Localization and Detection with Distance Estimation","date":"2025-04-11","arxiv_id":"2504.08644","repositories_listed":1,"syntology":null},{"url":"/paper/mvanet-multi-stage-video-attention-network","title":"MVANet: Multi-Stage Video Attention Network for Sound Event Localization and Detection with Source Distance Estimation","date":"2024-11-21","arxiv_id":"2411.14153","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-sound-event-localization-and","title":"Real-Time Sound Event Localization and Detection: Deployment Challenges on Edge Devices","date":"2024-09-18","arxiv_id":"2409.11700","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multi-target-tdoa-features-for-sound","title":"Learning Multi-Target TDOA Features for Sound Event Localization and Detection","date":"2024-08-30","arxiv_id":"2408.17166","repositories_listed":1,"syntology":null},{"url":"/paper/mff-einv2-multi-scale-feature-fusion-across","title":"MFF-EINV2: Multi-scale Feature Fusion across Spectral-Spatial-Temporal Domains for Sound Event Localization and Detection","date":"2024-06-13","arxiv_id":"2406.08771","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/enhanced-sound-event-localization-and","title":"Enhanced Sound Event Localization and Detection in Real 360-degree audio-visual soundscapes","date":"2024-01-29","arxiv_id":"2401.17129","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-scaper-a-library-to-simulate-and","title":"Spatial Scaper: A Library to Simulate and Augment Soundscapes for Sound Event Localization and Detection in Realistic Rooms","date":"2024-01-19","arxiv_id":"2401.12238","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_unverified":1,"n_pointer_only":12}},{"url":"/paper/selective-memory-meta-learning-with","title":"Selective-Memory Meta-Learning with Environment Representations for Sound Event Localization and Detection","date":"2023-12-27","arxiv_id":"2312.16422","repositories_listed":1,"syntology":null},{"url":"/paper/fusion-of-audio-and-visual-embeddings-for","title":"Fusion of Audio and Visual Embeddings for Sound Event Localization and Detection","date":"2023-12-14","arxiv_id":"2312.09034","repositories_listed":1,"syntology":null},{"url":"/paper/w2v-seld-a-sound-event-localization-and","title":"w2v-SELD: A Sound Event Localization and Detection Framework for Self-Supervised Spatial Audio Pre-Training","date":"2023-12-12","arxiv_id":"2312.06907","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-geometrical-acoustic-simulations","title":"Leveraging Geometrical Acoustic Simulations of Spatial Room Impulse Responses for Improved Sound Event Detection and Localization","date":"2023-09-06","arxiv_id":"2309.03337","repositories_listed":1,"syntology":null},{"url":"/paper/starss23-an-audio-visual-dataset-of-spatial","title":"STARSS23: An Audio-Visual Dataset of Spatial Recordings of Real Scenes with Spatiotemporal Annotations of Sound Events","date":"2023-06-15","arxiv_id":"2306.09126","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/perception-test-a-diagnostic-benchmark-for-2","title":"Perception Test: A Diagnostic Benchmark for Multimodal Video Models","date":"2023-05-23","arxiv_id":"2305.13786","repositories_listed":1,"syntology":null},{"url":"/paper/ad-yolo-you-look-only-once-in-training","title":"AD-YOLO: You Look Only Once in Training Multiple Sound Event Localization and Detection","date":"2023-03-28","arxiv_id":"2303.15703","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/a-synapse-threshold-synergistic-learning","title":"A Synapse-Threshold Synergistic Learning Approach for Spiking Neural Networks","date":"2022-06-10","arxiv_id":"2206.06129","repositories_listed":1,"syntology":null},{"url":"/paper/dual-quaternion-ambisonics-array-for-six","title":"Dual Quaternion Ambisonics Array for Six-Degree-of-Freedom Acoustic Representation","date":"2022-04-04","arxiv_id":"2204.01851","repositories_listed":1,"syntology":null},{"url":"/paper/filler-word-detection-and-classification-a","title":"Filler Word Detection and Classification: A Dataset and Benchmark","date":"2022-03-28","arxiv_id":"2203.15135","repositories_listed":1,"syntology":null},{"url":"/paper/l3das22-challenge-learning-3d-audio-sources","title":"L3DAS22 Challenge: Learning 3D Audio Sources in a Real Office Environment","date":"2022-02-21","arxiv_id":"2202.10372","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/echo-aware-adaptation-of-sound-event","title":"Echo-aware Adaptation of Sound Event Localization and Detection in Unknown Environments","date":"2022-02-18","arxiv_id":"2202.09121","repositories_listed":1,"syntology":null},{"url":"/paper/wearable-seld-dataset-dataset-for-sound-event","title":"Wearable SELD dataset: Dataset for sound event localization and detection using wearable devices around head","date":"2022-02-17","arxiv_id":"2202.08458","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-mixup-directional-loudness","title":"Spatial mixup: Directional loudness modification as data augmentation for sound event localization and detection","date":"2021-10-12","arxiv_id":"2110.06126","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}