{"url":"/dataset/unav-100","name":"UnAV-100","full_name":null,"description_markdown":"Existing audio-visual event localization (AVE) handles manually trimmed videos with only a single instance in each of them. However, this setting is unrealistic as natural videos often contain numerous audio-visual events with different categories. To better adapt to real-life applications, we focus on the task of dense-localizing audio-visual events, which aims to jointly localize and recognize all audio-visual events occurring in an untrimmed video. To tackle this problem, we introduce the first Untrimmed Audio-Visual (UnAV-100) dataset, which contains 10K untrimmed videos with over 30K audio-visual events covering 100 event categories. Each video has 2.8 audio-visual events on average, and the events are usually related to each other and might co-occur as in real-life scenes. We believe our UnAV-100, with its realistic complexity, can promote the exploration on comprehensive audio-visual video understanding.","description_withheld":null,"homepage":"https://unav100.github.io","introduced_date":"2023-03-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/dense-localizing-audio-visual-events-in","title":"Dense-Localizing Audio-Visual Events in Untrimmed Videos: A Large-Scale Benchmark and Baseline","first_author":"Tiantian Geng","url":null},"license":{"name":"https://creativecommons.org/licenses/by/4.0/","url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"audio-visual event localization","url":"/task/audio-visual-event-localization","datasets_with_task":"/datasets/task/audio-visual-event-localization"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["UnAV-100"],"data_loaders":[],"num_papers_in_archive":13,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/audio-visual-event-localization-on-unav-100","task":"audio-visual event localization","dataset_variant":"UnAV-100","rows":2,"metrics":[" mAP","AP@IOU0.5"],"first_row_in_archive_order":{"model":"UnAV","paper":"/paper/dense-localizing-audio-visual-events-in","metrics":{" mAP":"47.8","AP@IOU0.5":"50.6"},"code_links":[{"title":"ttgeng233/UnAV","url":"https://github.com/ttgeng233/UnAV"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/dense-localizing-audio-visual-events-in","title":"Dense-Localizing Audio-Visual Events in Untrimmed Videos: A Large-Scale Benchmark and Baseline","date":"2023-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":6,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/actionformer-localizing-moments-of-actions","title":"ActionFormer: Localizing Moments of Actions with Transformers","date":"2022-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":13,"samples_ran":6,"samples_unverified":7,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}