{"url":"/dataset/grid","name":"GRID Dataset","full_name":null,"description_markdown":"The QMUL underGround Re-IDentification (**GRID**) dataset contains 250 pedestrian image pairs. Each pair contains two images of the same individual seen from different camera views. All images are captured from 8 disjoint camera views installed in a busy underground station. The figures beside show a snapshot of each of the camera views of the station and sample images in the dataset. The dataset is challenging due to variations of pose, colours, lighting changes; as well as poor image quality caused by low spatial resolution.\r\n\r\nSource: [https://personal.ie.cuhk.edu.hk/~ccloy/downloads_qmul_underground_reid.html](https://personal.ie.cuhk.edu.hk/~ccloy/downloads_qmul_underground_reid.html)","description_withheld":null,"homepage":"https://personal.ie.cuhk.edu.hk/~ccloy/downloads_qmul_underground_reid.html","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"task","url":null,"datasets_with_task":"/datasets/task/task"},{"name":"Speech Enhancement","url":"/task/speech-enhancement","datasets_with_task":"/datasets/task/speech-enhancement"},{"name":"Speech Separation","url":"/task/speech-separation","datasets_with_task":"/datasets/task/speech-separation"},{"name":"Lipreading","url":"/task/lipreading","datasets_with_task":"/datasets/task/lipreading"},{"name":"Speaker-Specific Lip to Speech Synthesis","url":"/task/speaker-specific-lip-to-speech-synthesis","datasets_with_task":"/datasets/task/speaker-specific-lip-to-speech-synthesis"},{"name":"Lip Reading","url":"/task/lip-reading","datasets_with_task":"/datasets/task/lip-reading"}],"languages":[{"name":"Italian","url":"/datasets/language/italian"}],"variants":["GRID corpus (mixed-speech)","GRID Dataset"],"data_loaders":[],"num_papers_in_archive":10,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/lipreading-on-grid-corpus-mixed-speech","task":"Lipreading","dataset_variant":"GRID corpus (mixed-speech)","rows":5,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"CTC/Attention","paper":"/paper/visual-speech-recognition-for-multiple","metrics":{"Word Error Rate (WER)":"1.2"},"code_links":[{"title":"mpc001/Visual_Speech_Recognition_for_Multiple_Languages","url":"https://github.com/mpc001/Visual_Speech_Recognition_for_Multiple_Languages"},{"title":"david-gimeno/lip-rtve","url":"https://github.com/david-gimeno/lip-rtve"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speaker-specific-lip-to-speech-synthesis-on","task":"Speaker-Specific Lip to Speech Synthesis","dataset_variant":"GRID corpus (mixed-speech)","rows":2,"metrics":["ESTOI","PESQ","STOI"],"first_row_in_archive_order":{"model":"Visual Voice Memory","paper":"/paper/speech-reconstruction-with-reminiscent-sound","metrics":{"ESTOI":"0.579","PESQ":"1.984","STOI":"0.738"},"code_links":[{"title":"joannahong/Speech-Reconstruction-with-Reminiscent-Sound-via-Visual-Voice-Memory","url":"https://github.com/joannahong/Speech-Reconstruction-with-Reminiscent-Sound-via-Visual-Voice-Memory"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-separation-on-grid-corpus-mixed-speech","task":"Speech Separation","dataset_variant":"GRID corpus (mixed-speech)","rows":2,"metrics":["SDR"],"first_row_in_archive_order":{"model":"","paper":"/paper/midi-multi-instance-diffusion-for-single","metrics":{"SDR":"9.6"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/lip-reading-on-grid-corpus-mixed-speech","task":"Lip Reading","dataset_variant":"GRID corpus (mixed-speech)","rows":1,"metrics":["WER"],"first_row_in_archive_order":{"model":"Lip2Wav","paper":"/paper/learning-individual-speaking-styles-for","metrics":{"WER":"14.08"},"code_links":[{"title":"Rudrabha/Lip2Wav","url":"https://github.com/Rudrabha/Lip2Wav"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-enhancement-on-grid-corpus-mixed","task":"Speech Enhancement","dataset_variant":"GRID corpus (mixed-speech)","rows":1,"metrics":["PESQ"],"first_row_in_archive_order":{"model":"Audio-Visual concat-ref","paper":"/paper/face-landmark-based-speaker-independent-audio","metrics":{"PESQ":"2.70"},"code_links":[{"title":"dr-pato/audio_visual_speech_enhancement","url":"https://github.com/dr-pato/audio_visual_speech_enhancement"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/midi-multi-instance-diffusion-for-single","title":"MIDI: Multi-Instance Diffusion for Single Image to 3D Scene Generation","date":"2024-12-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/visual-speech-recognition-for-multiple","title":"Visual Speech Recognition for Multiple Languages in the Wild","date":"2022-02-26","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/speech-reconstruction-with-reminiscent-sound","title":"Speech Reconstruction with Reminiscent Sound via Visual Voice Memory","date":"2021-11-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-individual-speaking-styles-for","title":"Learning Individual Speaking Styles for Accurate Lip to Speech Synthesis","date":"2020-05-17","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/can-we-read-speech-beyond-the-lips-rethinking","title":"Can We Read Speech Beyond the Lips? Rethinking RoI Selection for Deep Visual Speech Recognition","date":"2020-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/face-landmark-based-speaker-independent-audio","title":"Face Landmark-based Speaker-Independent Audio-Visual Speech Enhancement in Multi-Talker Environments","date":"2018-11-06","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/lcanet-end-to-end-lipreading-with-cascaded","title":"LCANet: End-to-End Lipreading with Cascaded Attention-CTC","date":"2018-03-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lip-reading-sentences-in-the-wild","title":"Lip Reading Sentences in the Wild","date":"2016-11-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lipnet-end-to-end-sentence-level-lipreading","title":"LipNet: End-to-End Sentence-level Lipreading","date":"2016-11-05","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":0,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":24,"samples_ran":1,"samples_unverified":23,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}