{"url":"/task/active-speaker-detection","name":"Active Speaker Detection","slug":"active-speaker-detection","description_markdown":null,"categories":[{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":63,"papers_with_code":30,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/active-speaker-detection-on-lrs3-ted","slug":"active-speaker-detection-on-lrs3-ted","dataset":"LRS3-TED","dataset_url":"/dataset/lrs3-ted","rows_in_archive":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GestSync","paper_title":"GestSync: Determining who is speaking without a talking head","paper_url":"/paper/gestsync-determining-who-is-speaking-without","paper_date":"2023-10-08","arxiv_id":"2310.05304","code_links":[{"title":"Sindhu-Hegde/gestsync","url":"https://github.com/Sindhu-Hegde/gestsync"}],"syntology":null}}],"datasets":[{"url":"/dataset/lrs3-ted","name":"LRS3-TED","full_name":"","num_papers_in_archive":63},{"url":"/dataset/unitalk","name":"UniTalk","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/fraud-detection","name":"Fraud Detection"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":30,"tagged_in_all":63,"items":[{"url":"/paper/is-someone-speaking-exploring-long-term","title":"Is Someone Speaking? Exploring Long-term Temporal Features for Audio-visual Active Speaker Detection","date":"2021-07-14","arxiv_id":"2107.06592","repositories_listed":4,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/end-to-end-active-speaker-detection","title":"End-to-End Active Speaker Detection","date":"2022-03-27","arxiv_id":"2203.14250","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/loconet-long-short-context-network-for-active","title":"LoCoNet: Long-Short Context Network for Active Speaker Detection","date":"2023-01-19","arxiv_id":"2301.08237","repositories_listed":2,"syntology":null},{"url":"/paper/learning-long-term-spatial-temporal-graphs","title":"Learning Long-Term Spatial-Temporal Graphs for Active Speaker Detection","date":"2022-07-15","arxiv_id":"2207.07783","repositories_listed":2,"syntology":null},{"url":"/paper/ava-activespeaker-an-audio-visual-dataset-for","title":"AVA-ActiveSpeaker: An Audio-Visual Dataset for Active Speaker Detection","date":"2019-01-05","arxiv_id":"1901.01342","repositories_listed":2,"syntology":null},{"url":"/paper/unitalk-towards-universal-active-speaker","title":"UniTalk: Towards Universal Active Speaker Detection in Real World Scenarios","date":"2025-05-28","arxiv_id":"2505.21954","repositories_listed":1,"syntology":null},{"url":"/paper/cogenav-versatile-audio-visual-representation","title":"CoGenAV: Versatile Audio-Visual Representation Learning via Contrastive-Generative Synchronization","date":"2025-05-06","arxiv_id":"2505.03186","repositories_listed":1,"syntology":null},{"url":"/paper/laser-lip-landmark-assisted-speaker-detection","title":"LASER: Lip Landmark Assisted Speaker Detection for Robustness","date":"2025-01-21","arxiv_id":"2501.11899","repositories_listed":1,"syntology":null},{"url":"/paper/asdnb-merging-face-with-body-cues-for-robust","title":"ASDnB: Merging Face with Body Cues For Robust Active Speaker Detection","date":"2024-12-11","arxiv_id":"2412.08594","repositories_listed":1,"syntology":null},{"url":"/paper/bias-a-body-based-interpretable-active","title":"BIAS: A Body-based Interpretable Active Speaker Approach","date":"2024-12-06","arxiv_id":"2412.05150","repositories_listed":1,"syntology":null},{"url":"/paper/fabulight-asd-unveiling-speech-activity-via","title":"FabuLight-ASD: Unveiling Speech Activity via Body Language","date":"2024-11-20","arxiv_id":"2411.13674","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-of-human-motion-achieves-natural","title":"Imitation of human motion achieves natural head movements for humanoid robots in an active-speaker detection task","date":"2024-07-16","arxiv_id":"2407.11915","repositories_listed":1,"syntology":null},{"url":"/paper/annotheia-a-semi-automatic-annotation-toolkit","title":"AnnoTheia: A Semi-Automatic Annotation Toolkit for Audio-Visual Speech Technologies","date":"2024-02-20","arxiv_id":"2402.13152","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-visual-supervision-for-array-based","title":"Leveraging Visual Supervision for Array-based Active Speaker Detection and Localization","date":"2023-12-21","arxiv_id":"2312.14021","repositories_listed":1,"syntology":null},{"url":"/paper/gestsync-determining-who-is-speaking-without","title":"GestSync: Determining who is speaking without a talking head","date":"2023-10-08","arxiv_id":"2310.05304","repositories_listed":1,"syntology":null},{"url":"/paper/talknce-improving-active-speaker-detection","title":"TalkNCE: Improving Active Speaker Detection with Talk-Aware Contrastive Learning","date":"2023-09-21","arxiv_id":"2309.12306","repositories_listed":1,"syntology":null},{"url":"/paper/target-active-speaker-detection-with-audio","title":"Target Active Speaker Detection with Audio-visual Cues","date":"2023-05-22","arxiv_id":"2305.12831","repositories_listed":1,"syntology":null},{"url":"/paper/wasd-a-wilder-active-speaker-detection","title":"WASD: A Wilder Active Speaker Detection Dataset","date":"2023-03-09","arxiv_id":"2303.05321","repositories_listed":1,"syntology":null},{"url":"/paper/a-light-weight-model-for-active-speaker","title":"A Light Weight Model for Active Speaker Detection","date":"2023-03-08","arxiv_id":"2303.04439","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/audio-visual-activity-guided-cross-modal","title":"Audio-Visual Activity Guided Cross-Modal Identity Association for Active Speaker Detection","date":"2022-12-01","arxiv_id":"2212.00539","repositories_listed":1,"syntology":null},{"url":"/paper/whose-emotion-matters-speaker-detection","title":"Whose Emotion Matters? Speaking Activity Localisation without Prior Knowledge","date":"2022-11-23","arxiv_id":"2211.15377","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-active-speaker-detection-in","title":"Unsupervised active speaker detection in media content using cross-modal information","date":"2022-09-24","arxiv_id":"2209.11896","repositories_listed":1,"syntology":null},{"url":"/paper/look-listen-multi-modal-correlation-learning","title":"Look\\&Listen: Multi-Modal Correlation Learning for Active Speaker Detection and Speech Enhancement","date":"2022-03-04","arxiv_id":"2203.02216","repositories_listed":1,"syntology":null},{"url":"/paper/look-who-s-talking-active-speaker-detection","title":"Look Who's Talking: Active Speaker Detection in the Wild","date":"2021-08-17","arxiv_id":"2108.07640","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-design-a-three-stage-architecture-for","title":"How to Design a Three-Stage Architecture for Audio-Visual Active Speaker Detection in the Wild","date":"2021-06-07","arxiv_id":"2106.03932","repositories_listed":1,"syntology":null},{"url":"/paper/nus-hlt-report-for-activitynet-challenge-2021","title":"NUS-HLT Report for ActivityNet Challenge 2021 AVA (Speaker)","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/maas-multi-modal-assignation-for-active","title":"MAAS: Multi-modal Assignation for Active Speaker Detection","date":"2021-01-11","arxiv_id":"2101.03682","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-of-audio-visual","title":"Self-Supervised Learning of Audio-Visual Objects from Video","date":"2020-08-10","arxiv_id":"2008.04237","repositories_listed":1,"syntology":{"n":14,"n_ran":1,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/active-speakers-in-context","title":"Active Speakers in Context","date":"2020-05-20","arxiv_id":"2005.09812","repositories_listed":1,"syntology":null},{"url":"/paper/bio-inspired-modality-fusion-for-active","title":"Bio-Inspired Modality Fusion for Active Speaker Detection","date":"2020-02-28","arxiv_id":"2003.00063","repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}