{"url":"/dataset/lrw-1000","name":"CAS-VSR-W1k (LRW-1000)","full_name":"CAS-VSR-W1k (LRW-1000)","description_markdown":"*LRW-1000 has been renamed as CAS-VSR-W1k.** It is a naturally-distributed large-scale benchmark for word-level lipreading in the wild, including 1000 classes with about 718,018 video samples from more than 2000 individual speakers. There are more than 1,000,000 Chinese character instances in total. Each class corresponds to the syllables of a Mandarin word which is composed by one or several Chinese characters. This dataset aims to cover a natural variability over different speech modes and imaging conditions to incorporate challenges encountered in practical applications.\r\n\r\nSource: [VIPL](https://vipl.ict.ac.cn/en/view_database.php?id=13)\r\nImage Source: [https://arxiv.org/pdf/1810.06990v6.pdf](https://arxiv.org/pdf/1810.06990v6.pdf)","description_withheld":null,"homepage":"https://vipl.ict.ac.cn/en/view_database.php?id=13","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/lrw-1000-a-naturally-distributed-large-scale","title":"LRW-1000: A Naturally-Distributed Large-Scale Benchmark for Lip Reading in the Wild","first_author":"Shuang Yang","url":null},"license":{"name":"research-only, non-commercial","url":"http://vipl.ict.ac.cn/resources/datasets/LRW-1000/LRW-1000-Release%20Agreement.pdf"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Visual Speech Recognition","url":"/task/visual-speech-recognition","datasets_with_task":"/datasets/task/visual-speech-recognition"},{"name":"Lipreading","url":"/task/lipreading","datasets_with_task":"/datasets/task/lipreading"},{"name":"Audio-Visual Speech Recognition","url":"/task/audio-visual-speech-recognition","datasets_with_task":"/datasets/task/audio-visual-speech-recognition"},{"name":"Lip Reading","url":"/task/lip-reading","datasets_with_task":"/datasets/task/lip-reading"}],"languages":[],"variants":["CAS-VSR-W1k (LRW-1000)"],"data_loaders":[],"num_papers_in_archive":9,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/lipreading-on-lrw-1000","task":"Lipreading","dataset_variant":"CAS-VSR-W1k (LRW-1000)","rows":9,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"SyncVSR (Word Boundary)","paper":"/paper/syncvsr-data-efficient-visual-speech","metrics":{"Top-1 Accuracy":"58.2"},"code_links":[{"title":"KAIST-AILab/SyncVSR","url":"https://github.com/KAIST-AILab/SyncVSR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/syncvsr-data-efficient-visual-speech","title":"SyncVSR: Data-Efficient Visual Speech Recognition with End-to-End Crossmodal Audio Token Synchronization","date":"2024-06-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-modality-associative-bridging-through-1","title":"Multi-modality Associative Bridging through Memory: Speech Sound Recollected from Face Video","date":"2022-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distinguishing-homophenes-using-multi-head-1","title":"Distinguishing Homophenes Using Multi-Head Visual-Audio Memory for Lip Reading","date":"2022-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learn-an-effective-lip-reading-model-without","title":"Learn an Effective Lip Reading Model without Pains","date":"2020-11-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/mutual-information-maximization-for-effective","title":"Mutual Information Maximization for Effective Lip Reading","date":"2020-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deformation-flow-based-two-stream-network-for","title":"Deformation Flow Based Two-Stream Network for Lip Reading","date":"2020-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pseudo-convolutional-policy-gradient-for","title":"Pseudo-Convolutional Policy Gradient for Sequence-to-Sequence Lip-Reading","date":"2020-03-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/can-we-read-speech-beyond-the-lips-rethinking","title":"Can We Read Speech Beyond the Lips? Rethinking RoI Selection for Deep Visual Speech Recognition","date":"2020-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}