{"url":"/dataset/lrw","name":"LRW","full_name":"Lip Reading in the Wild","description_markdown":"The **Lip Reading in the Wild** (**LRW**) dataset  a large-scale audio-visual database that contains 500 different words from over 1,000 speakers. Each utterance has 29 frames, whose boundary is centered around the target word. The database is divided into training, validation and test sets. The training set contains at least 800 utterances for each class while the validation and test sets contain 50 utterances.\r\n\r\nSource: [Towards Pose-invariant Lip-Reading](https://arxiv.org/abs/1911.06095)\r\nImage Source: [https://www.robots.ox.ac.uk/~vgg/data/lip_reading/lrw1.html](https://www.robots.ox.ac.uk/~vgg/data/lip_reading/lrw1.html)","description_withheld":null,"homepage":"https://www.robots.ox.ac.uk/~vgg/data/lip_reading/lrw1.html","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"Lip Reading in the Wild","first_author":null,"url":"https://doi.org/10.1007/978-3-319-54184-6_6"},"license":{"name":"Custom (research-only, non-commercial, attribution)","url":"https://www.bbc.co.uk/rd/projects/lip-reading-datasets"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Lipreading","url":"/task/lipreading","datasets_with_task":"/datasets/task/lipreading"},{"name":"Lip to Speech Synthesis","url":"/task/lip-to-speech-synthesis","datasets_with_task":"/datasets/task/lip-to-speech-synthesis"},{"name":"Audio-Visual Speech Recognition","url":"/task/audio-visual-speech-recognition","datasets_with_task":"/datasets/task/audio-visual-speech-recognition"},{"name":"Lip Reading","url":"/task/lip-reading","datasets_with_task":"/datasets/task/lip-reading"},{"name":"Unconstrained Lip-synchronization","url":"/task/lip-sync","datasets_with_task":"/datasets/task/lip-sync"},{"name":"Visual Keyword Spotting","url":"/task/visual-keyword-spotting","datasets_with_task":"/datasets/task/visual-keyword-spotting"},{"name":"Talking Face Generation","url":"/task/talking-face-generation","datasets_with_task":"/datasets/task/talking-face-generation"},{"name":"Landmark-based Lipreading","url":"/task/landmark-based-lipreading","datasets_with_task":"/datasets/task/landmark-based-lipreading"}],"languages":[],"variants":["LRW","Lip Reading in the Wild","Lipreading in the Wild"],"data_loaders":[{"repo":"https://github.com/Rudrabha/Wav2Lip","url":"https://arxiv.org/abs/2008.10010","frameworks":["pytorch"]}],"num_papers_in_archive":188,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/lipreading-on-lip-reading-in-the-wild","task":"Lipreading","dataset_variant":"Lip Reading in the Wild","rows":22,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"SyncVSR (Word Boundary)","paper":"/paper/syncvsr-data-efficient-visual-speech","metrics":{"Top-1 Accuracy":"95.0"},"code_links":[{"title":"KAIST-AILab/SyncVSR","url":"https://github.com/KAIST-AILab/SyncVSR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/landmark-based-lipreading-on-lrw","task":"Landmark-based Lipreading","dataset_variant":"LRW","rows":5,"metrics":["Top 1 Accuracy"],"first_row_in_archive_order":{"model":"SyncVSR (Word Boundary)","paper":"/paper/syncvsr-data-efficient-visual-speech","metrics":{"Top 1 Accuracy":"80.3"},"code_links":[{"title":"KAIST-AILab/SyncVSR","url":"https://github.com/KAIST-AILab/SyncVSR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/audio-visual-speech-recognition-on-lrw","task":"Audio-Visual Speech Recognition","dataset_variant":"LRW","rows":3,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"AVCRFormer","paper":"/paper/audio-visual-speech-recognition-based-on","metrics":{"Top-1 Accuracy":"98.81"},"code_links":[{"title":"SMIL-SPCRAS/AVCRFormer","url":"https://github.com/SMIL-SPCRAS/AVCRFormer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/lip-sync-on-lrw","task":"Unconstrained Lip-synchronization","dataset_variant":"LRW","rows":2,"metrics":["FID","LSE-C","LSE-D"],"first_row_in_archive_order":{"model":"Wav2Lip + GAN","paper":"/paper/a-lip-sync-expert-is-all-you-need-for-speech","metrics":{"FID":"2.475","LSE-C":"7.263","LSE-D":"6.774"},"code_links":[{"title":"Rudrabha/Wav2Lip","url":"https://github.com/Rudrabha/Wav2Lip"},{"title":"mowshon/lipsync","url":"https://github.com/mowshon/lipsync"},{"title":"PrashanthaTP/wav2mov","url":"https://github.com/PrashanthaTP/wav2mov"},{"title":"rockstar-0000/lip_sync_test","url":"https://github.com/rockstar-0000/lip_sync_test"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/lip-reading-on-lrw","task":"Lip Reading","dataset_variant":"LRW","rows":1,"metrics":["WER"],"first_row_in_archive_order":{"model":"Lip2Wav","paper":"/paper/learning-individual-speaking-styles-for","metrics":{"WER":"34.2"},"code_links":[{"title":"Rudrabha/Lip2Wav","url":"https://github.com/Rudrabha/Lip2Wav"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/lip-to-speech-synthesis-on-lrw","task":"Lip to Speech Synthesis","dataset_variant":"LRW","rows":1,"metrics":["ESTOI","PESQ","STOI"],"first_row_in_archive_order":{"model":"Lip2Wav","paper":"/paper/learning-individual-speaking-styles-for","metrics":{"ESTOI":"0.344","PESQ":"1.197","STOI":"0.543"},"code_links":[{"title":"Rudrabha/Lip2Wav","url":"https://github.com/Rudrabha/Lip2Wav"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/talking-face-generation-on-lrw","task":"Talking Face Generation","dataset_variant":"LRW","rows":1,"metrics":["LMD","SSIM"],"first_row_in_archive_order":{"model":"LipGAN","paper":"/paper/towards-automatic-face-to-face-translation-1","metrics":{"LMD":"0.60","SSIM":"0.96"},"code_links":[{"title":"Rudrabha/LipGAN","url":"https://github.com/Rudrabha/LipGAN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-keyword-spotting-on-lrw","task":"Visual Keyword Spotting","dataset_variant":"LRW","rows":1,"metrics":["Top-1 Accuracy","Top-5 Accuracy","mAP"],"first_row_in_archive_order":{"model":"Transpotter","paper":"/paper/visual-keyword-spotting-with-attention","metrics":{"Top-1 Accuracy":"85.8","Top-5 Accuracy":"99.6","mAP":"64.1"},"code_links":[{"title":"prajwalkr/transpotter","url":"https://github.com/prajwalkr/transpotter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/syncvsr-data-efficient-visual-speech","title":"SyncVSR: Data-Efficient Visual Speech Recognition with End-to-End Crossmodal Audio Token Synchronization","date":"2024-06-18","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/audio-visual-speech-recognition-based-on","title":"Audio-Visual Speech Recognition based on Regulated Transformer and Spatio-Temporal Fusion Strategy for Driver Assistive Systems","date":"2024-05-09","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/another-point-of-view-on-visual-speech","title":"Another Point of View on Visual Speech Recognition","date":"2023-08-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/audio-visual-speech-and-gesture-recognition","title":"Audio-Visual Speech and Gesture Recognition by Sensors of Mobile Devices","date":"2023-02-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/training-strategies-for-improved-lip-reading","title":"Training Strategies for Improved Lip-reading","date":"2022-09-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/visual-speech-recognition-in-a-driver","title":"Visual Speech Recognition in a Driver Assistance System","date":"2022-08-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/accurate-and-resource-efficient-lipreading","title":"Accurate and Resource-Efficient Lipreading with Efficientnetv2 and Transformers","date":"2022-05-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-modality-associative-bridging-through-1","title":"Multi-modality Associative Bridging through Memory: Speech Sound Recollected from Face Video","date":"2022-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distinguishing-homophenes-using-multi-head-1","title":"Distinguishing Homophenes Using Multi-Head Visual-Audio Memory for Lip Reading","date":"2022-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/leveraging-uni-modal-self-supervised-learning-1","title":"Leveraging Unimodal Self-Supervised Learning for Multimodal Audio-Visual Speech Recognition","date":"2022-02-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visual-keyword-spotting-with-attention","title":"Visual Keyword Spotting with Attention","date":"2021-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adaptive-semantic-spatio-temporal-graph","title":"Adaptive Semantic-Spatio-Temporal Graph Convolutional Network for Lip Reading","date":"2021-08-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/part-based-lipreading-for-audio-visual-speech","title":"Part-based Lipreading for Audio-Visual Speech Recognition","date":"2020-12-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learn-an-effective-lip-reading-model-without","title":"Learn an Effective Lip Reading Model without Pains","date":"2020-11-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/lip-graph-assisted-audio-visual-speech","title":"Lip Graph Assisted Audio-Visual Speech Recognition Using Bidirectional Synchronous Fusion","date":"2020-10-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-lip-sync-expert-is-all-you-need-for-speech","title":"A Lip Sync Expert Is All You Need for Speech to Lip Generation In The Wild","date":"2020-08-23","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-practical-lipreading-with-distilled","title":"Towards Practical Lipreading with Distilled and Efficient Models","date":"2020-07-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/spotfast-networks-with-memory-augmented","title":"SpotFast Networks with Memory Augmented Lateral Transformers for Lipreading","date":"2020-05-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-individual-speaking-styles-for","title":"Learning Individual Speaking Styles for Accurate Lip to Speech Synthesis","date":"2020-05-17","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/discriminative-multi-modality-speech","title":"Discriminative Multi-modality Speech Recognition","date":"2020-05-12","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/mutual-information-maximization-for-effective","title":"Mutual Information Maximization for Effective Lip Reading","date":"2020-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deformation-flow-based-two-stream-network-for","title":"Deformation Flow Based Two-Stream Network for Lip Reading","date":"2020-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pseudo-convolutional-policy-gradient-for","title":"Pseudo-Convolutional Policy Gradient for Sequence-to-Sequence Lip-Reading","date":"2020-03-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/can-we-read-speech-beyond-the-lips-rethinking","title":"Can We Read Speech Beyond the Lips? Rethinking RoI Selection for Deep Visual Speech Recognition","date":"2020-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/towards-automatic-face-to-face-translation-1","title":"Towards Automatic Face-to-Face Translation","date":"2020-03-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lipreading-using-temporal-convolutional","title":"Lipreading using Temporal Convolutional Networks","date":"2020-01-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-grained-spatio-temporal-modeling-for","title":"Multi-Grained Spatio-temporal Modeling for Lip-reading","date":"2019-08-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-audiovisual-speech-recognition","title":"End-to-end Audiovisual Speech Recognition","date":"2018-02-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/combining-residual-networks-with-lstms-for","title":"Combining Residual Networks with LSTMs for Lipreading","date":"2017-03-12","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":24,"samples_ran":8,"samples_unverified":16,"pointer_only_for_licence":6,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}