{"url":"/dataset/ava-speech","name":"AVA-Speech","full_name":null,"description_markdown":"Contains densely labeled speech activity in YouTube videos, with the goal of creating a shared, available dataset for this task. \r\n\r\nSource: [AVA-Speech: A Densely Labeled Dataset of Speech Activity in Movies](/paper/ava-speech-a-densely-labeled-dataset-of)","description_withheld":null,"homepage":"https://research.google.com/ava/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/ava-speech-a-densely-labeled-dataset-of","title":"AVA-Speech: A Densely Labeled Dataset of Speech Activity in Movies","first_author":null,"url":null},"license":null,"modalities":[],"tasks":[{"name":"Action Detection","url":"/task/action-detection","datasets_with_task":"/datasets/task/action-detection"},{"name":"Speaker Diarization","url":"/task/speaker-diarization","datasets_with_task":"/datasets/task/speaker-diarization"},{"name":"Activity Detection","url":"/task/activity-detection","datasets_with_task":"/datasets/task/activity-detection"}],"languages":[],"variants":["AVA-Speech"],"data_loaders":[],"num_papers_in_archive":10,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/activity-detection-on-ava-speech","task":"Activity Detection","dataset_variant":"AVA-Speech","rows":4,"metrics":["ROC-AUC"],"first_row_in_archive_order":{"model":"CNN-BiLSTM_best","paper":"/paper/a-hybrid-cnn-bilstm-voice-activity-detector","metrics":{"ROC-AUC":"95.14"},"code_links":[{"title":"NickWilkinson37/voxseg","url":"https://github.com/NickWilkinson37/voxseg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/sg-vad-stochastic-gates-based-speech-activity","title":"SG-VAD: Stochastic Gates Based Speech Activity Detection","date":"2022-10-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ada-vad-unpaired-adversarial-domain","title":"ADA-VAD: Unpaired Adversarial Domain Adaptation for Noise-Robust Voice Activity Detection","date":"2022-04-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-hybrid-cnn-bilstm-voice-activity-detector","title":"A Hybrid CNN-BiLSTM Voice Activity Detector","date":"2021-03-05","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}