{"url":"/dataset/lrs3-ted","name":"LRS3-TED","full_name":null,"description_markdown":"LRS3-TED is a multi-modal dataset for visual and audio-visual speech recognition. It includes face tracks from over 400 hours of TED and TEDx videos, along with the corresponding subtitles and word alignment boundaries. The new dataset is substantially larger in scale compared to other public datasets that are available for general research. \r\n\r\nSource: [LRS3-TED: a large-scale dataset for visual speech recognition](https://arxiv.org/pdf/1809.00496v2.pdf)","description_withheld":null,"homepage":"https://www.robots.ox.ac.uk/~vgg/data/lip_reading/lrs3.html","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/lrs3-ted-a-large-scale-dataset-for-visual","title":"LRS3-TED: a large-scale dataset for visual speech recognition","first_author":"Triantafyllos Afouras","url":null},"license":{"name":"Creative Commons BY-NC-ND 4.0 license","url":"https://creativecommons.org/licenses/by-nc-nd/4.0/legalcode"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Automatic Speech Recognition (ASR)","url":"/task/automatic-speech-recognition","datasets_with_task":"/datasets/task/automatic-speech-recognition"},{"name":"Active Speaker Detection","url":"/task/active-speaker-detection","datasets_with_task":"/datasets/task/active-speaker-detection"},{"name":"Visual Speech Recognition","url":"/task/visual-speech-recognition","datasets_with_task":"/datasets/task/visual-speech-recognition"},{"name":"Lipreading","url":"/task/lipreading","datasets_with_task":"/datasets/task/lipreading"},{"name":"Audio-Visual Speech Recognition","url":"/task/audio-visual-speech-recognition","datasets_with_task":"/datasets/task/audio-visual-speech-recognition"},{"name":"Visual Keyword Spotting","url":"/task/visual-keyword-spotting","datasets_with_task":"/datasets/task/visual-keyword-spotting"},{"name":"Gesture Synchronization","url":"/task/gesture-synchronization","datasets_with_task":"/datasets/task/gesture-synchronization"}],"languages":[],"variants":["LRS3-TED"],"data_loaders":[],"num_papers_in_archive":63,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/lipreading-on-lrs3-ted","task":"Lipreading","dataset_variant":"LRS3-TED","rows":23,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"LP + Conformer","paper":"/paper/conformers-are-all-you-need-for-visual-speech","metrics":{"Word Error Rate (WER)":"12.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/audio-visual-speech-recognition-on-lrs3-ted","task":"Audio-Visual Speech Recognition","dataset_variant":"LRS3-TED","rows":12,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"MMS-LLaMA","paper":"/paper/mms-llama-efficient-llm-based-audio-visual-1","metrics":{"Word Error Rate (WER)":"0.74"},"code_links":[{"title":"JeongHun0716/MMS-LLaMA","url":"https://github.com/JeongHun0716/MMS-LLaMA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-lrs3-ted","task":"Speech Recognition","dataset_variant":"LRS3-TED","rows":4,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"Whisper","paper":"/paper/whisper-flamingo-integrating-visual-features","metrics":{"Word Error Rate (WER)":"0.68"},"code_links":[{"title":"roudimit/whisper-flamingo","url":"https://github.com/roudimit/whisper-flamingo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-speech-recognition-on-lrs3-ted","task":"Visual Speech Recognition","dataset_variant":"LRS3-TED","rows":3,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"CTC/Attention","paper":"/paper/auto-avsr-audio-visual-speech-recognition","metrics":{"Word Error Rate (WER)":"19.1"},"code_links":[{"title":"mpc001/auto_avsr","url":"https://github.com/mpc001/auto_avsr"},{"title":"umbertocappellazzo/llama-avsr","url":"https://github.com/umbertocappellazzo/llama-avsr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/automatic-speech-recognition-asr-on-lrs3-ted","task":"Automatic Speech Recognition (ASR)","dataset_variant":"LRS3-TED","rows":2,"metrics":["WER","Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"DistillAV","paper":"/paper/audio-visual-representation-learning-via","metrics":{"WER":"1.4"},"code_links":[{"title":"jxzhanggg/DistillAV","url":"https://github.com/jxzhanggg/DistillAV"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/active-speaker-detection-on-lrs3-ted","task":"Active Speaker Detection","dataset_variant":"LRS3-TED","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GestSync","paper":"/paper/gestsync-determining-who-is-speaking-without","metrics":{"Accuracy":"87 %"},"code_links":[{"title":"Sindhu-Hegde/gestsync","url":"https://github.com/Sindhu-Hegde/gestsync"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-keyword-spotting-on-lrs3-ted","task":"Visual Keyword Spotting","dataset_variant":"LRS3-TED","rows":1,"metrics":["Top-1 Accuracy","Top-5 Accuracy","mAP","mAP IOU@0.5"],"first_row_in_archive_order":{"model":"Transpotter","paper":"/paper/visual-keyword-spotting-with-attention","metrics":{"Top-1 Accuracy":"52","Top-5 Accuracy":"77.1","mAP":"55.4","mAP IOU@0.5":"53.6"},"code_links":[{"title":"prajwalkr/transpotter","url":"https://github.com/prajwalkr/transpotter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mms-llama-efficient-llm-based-audio-visual-1","title":"MMS-LLaMA: Efficient LLM-based Audio-Visual Speech Recognition with Minimal Multimodal Speech Tokens","date":"2025-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/zero-avsr-zero-shot-audio-visual-speech","title":"Zero-AVSR: Zero-Shot Audio-Visual Speech Recognition with LLMs by Learning Language-Agnostic Speech Representations","date":"2025-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/audio-visual-representation-learning-via","title":"Audio-Visual Representation Learning via Knowledge Distillation from Speech Foundation Models","date":"2025-02-09","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/unified-speech-recognition-a-single-model-for","title":"Unified Speech Recognition: A Single Model for Auditory, Visual, and Audiovisual Inputs","date":"2024-11-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-are-strong-audio-visual","title":"Large Language Models are Strong Audio-Visual Speech Recognition Learners","date":"2024-09-18","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":9,"samples_unverified":3,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/syncvsr-data-efficient-visual-speech","title":"SyncVSR: Data-Efficient Visual Speech Recognition with End-to-End Crossmodal Audio Token Synchronization","date":"2024-06-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/whisper-flamingo-integrating-visual-features","title":"Whisper-Flamingo: Integrating Visual Features into Whisper for Audio-Visual Speech Recognition and Translation","date":"2024-06-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":5,"samples_unverified":13,"pointer_only_for_licence":18,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/where-visual-speech-meets-language-vsp-llm","title":"Where Visual Speech Meets Language: VSP-LLM Framework for Efficient and Context-Aware Visual Speech Processing","date":"2024-02-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":8,"samples_unverified":4,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/es3-evolving-self-supervised-learning-of","title":"ES3: Evolving Self-Supervised Learning of Robust Audio-Visual Speech Representations","date":"2024-01-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/gestsync-determining-who-is-speaking-without","title":"GestSync: Determining who is speaking without a talking head","date":"2023-10-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/auto-avsr-audio-visual-speech-recognition","title":"Auto-AVSR: Audio-Visual Speech Recognition with Automatic Labels","date":"2023-03-25","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/conformers-are-all-you-need-for-visual-speech","title":"Conformers are All You Need for Visual Speech Recognition","date":"2023-02-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/jointly-learning-visual-and-auditory-speech","title":"Jointly Learning Visual and Auditory Speech Representations from Raw Data","date":"2022-12-12","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/relaxed-attention-for-transformer-models","title":"Relaxed Attention for Transformer Models","date":"2022-09-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/visual-speech-recognition-for-multiple","title":"Visual Speech Recognition for Multiple Languages in the Wild","date":"2022-02-26","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/robust-self-supervised-audio-visual-speech","title":"Robust Self-Supervised Audio-Visual Speech Recognition","date":"2022-01-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-audio-visual-speech-representation-1","title":"Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction","date":"2022-01-05","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/visual-keyword-spotting-with-attention","title":"Visual Keyword Spotting with Attention","date":"2021-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sub-word-level-lip-reading-with-visual","title":"Sub-word Level Lip Reading With Visual Attention","date":"2021-10-14","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-audio-visual-speech-recognition","title":"End-to-end Audio-visual Speech Recognition with Conformers","date":"2021-02-12","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/discriminative-multi-modality-speech","title":"Discriminative Multi-modality Speech Recognition","date":"2020-05-12","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/asr-is-all-you-need-cross-modal-distillation","title":"ASR is all you need: cross-modal distillation for lip reading","date":"2019-11-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/recurrent-neural-network-transducer-for-audio","title":"Recurrent Neural Network Transducer for Audio-Visual Speech Recognition","date":"2019-11-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/spatio-temporal-fusion-based-convolutional","title":"Spatio-Temporal Fusion Based Convolutional Sequence Learning for Lip Reading","date":"2019-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deep-audio-visual-speech-recognition","title":"Deep Audio-Visual Speech Recognition","date":"2018-09-06","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/large-scale-visual-speech-recognition","title":"Large-Scale Visual Speech Recognition","date":"2018-07-13","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":58,"samples_ran":23,"samples_unverified":35,"pointer_only_for_licence":52,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}