{"url":"/dataset/ava-activespeaker","name":"AVA-ActiveSpeaker","full_name":null,"description_markdown":"Contains temporally labeled face tracks in video, where each face instance is labeled as speaking or not, and whether the speech is audible. This dataset contains about 3.65 million human labeled frames or about 38.5 hours of face tracks, and the corresponding audio. \r\n\r\nSource: [AVA-ActiveSpeaker: An Audio-Visual Dataset for Active Speaker Detection](/paper/ava-activespeaker-an-audio-visual-dataset-for)","description_withheld":null,"homepage":"https://research.google.com/ava/download.html#ava_active_speaker_download","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/ava-activespeaker-an-audio-visual-dataset-for","title":"AVA-ActiveSpeaker: An Audio-Visual Dataset for Active Speaker Detection","first_author":"Joseph Roth","url":null},"license":null,"modalities":[],"tasks":[{"name":"Speech Enhancement","url":"/task/speech-enhancement","datasets_with_task":"/datasets/task/speech-enhancement"},{"name":"Speaker Diarization","url":"/task/speaker-diarization","datasets_with_task":"/datasets/task/speaker-diarization"},{"name":"Self-Supervised Learning","url":"/task/self-supervised-learning","datasets_with_task":"/datasets/task/self-supervised-learning"},{"name":"Audio-Visual Active Speaker Detection","url":"/task/audio-visual-active-speaker-detection","datasets_with_task":"/datasets/task/audio-visual-active-speaker-detection"}],"languages":[],"variants":["AVA-ActiveSpeaker"],"data_loaders":[],"num_papers_in_archive":22,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/audio-visual-active-speaker-detection-on-ava","task":"Audio-Visual Active Speaker Detection","dataset_variant":"AVA-ActiveSpeaker","rows":20,"metrics":["validation mean average precision"],"first_row_in_archive_order":{"model":"LoCoNet+TalkNCE","paper":"/paper/talknce-improving-active-speaker-detection","metrics":{"validation mean average precision":"95.5%"},"code_links":[{"title":"kaistmm/TalkNCE","url":"https://github.com/kaistmm/TalkNCE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/laser-lip-landmark-assisted-speaker-detection","title":"LASER: Lip Landmark Assisted Speaker Detection for Robustness","date":"2025-01-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/talknce-improving-active-speaker-detection","title":"TalkNCE: Improving Active Speaker Detection with Talk-Aware Contrastive Learning","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-light-weight-model-for-active-speaker","title":"A Light Weight Model for Active Speaker Detection","date":"2023-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/loconet-long-short-context-network-for-active","title":"LoCoNet: Long-Short Context Network for Active Speaker Detection","date":"2023-01-19","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/audio-visual-activity-guided-cross-modal","title":"Audio-Visual Activity Guided Cross-Modal Identity Association for Active Speaker Detection","date":"2022-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-long-term-spatial-temporal-graphs","title":"Learning Long-Term Spatial-Temporal Graphs for Active Speaker Detection","date":"2022-07-15","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/unicon-ictcas-ucas-submission-to-the-ava","title":"UniCon+: ICTCAS-UCAS Submission to the AVA-ActiveSpeaker Task at ActivityNet Challenge 2022","date":"2022-06-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-active-speaker-detection","title":"End-to-End Active Speaker Detection","date":"2022-03-27","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sub-word-level-lip-reading-with-visual","title":"Sub-word Level Lip Reading With Visual Attention","date":"2021-10-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unicon-unified-context-network-for-robust","title":"UniCon: Unified Context Network for Robust Active Speaker Detection","date":"2021-08-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/how-to-design-a-three-stage-architecture-for","title":"How to Design a Three-Stage Architecture for Audio-Visual Active Speaker Detection in the Wild","date":"2021-06-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/active-speaker-detection-as-a-multi-objective","title":"Active Speaker Detection as a Multi-Objective Optimization with Uncertainty-based Multimodal Fusion","date":"2021-06-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/nus-hlt-report-for-activitynet-challenge-2021","title":"NUS-HLT Report for ActivityNet Challenge 2021 AVA (Speaker)","date":"2021-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ictcas-ucas-tal-submission-to-the-ava","title":"ICTCAS-UCAS-TAL Submission to the AVA-ActiveSpeaker Task at ActivityNet Challenge 2021","date":"2021-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/maas-multi-modal-assignation-for-active","title":"MAAS: Multi-modal Assignation for Active Speaker Detection","date":"2021-01-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/active-speakers-in-context","title":"Active Speakers in Context","date":"2020-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/naver-at-activitynet-challenge-2019-task-b","title":"Naver at ActivityNet Challenge 2019 -- Task B Active Speaker Detection (AVA)","date":"2019-06-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-task-learning-for-audio-visual-active","title":"Multi-Task Learning for Audio Visual Active Speaker Detection","date":"2019-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":5,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}