{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/visual-speech-recognition-in-a-driver","title":"Visual Speech Recognition in a Driver Assistance System","arxiv_id":null,"date":"2022-08-29","proceeding":"30th European Signal Processing Conference (EUSIPCO) 2022 8","authors":["Denis Ivanko","Dmitry Ryumin","Alexey Kashevnik","Alexandr Axyonov","Alexey Karpov"],"abstract":"Visual speech recognition or automated lipreading is a field of growing attention. Video data proved its usefulness in multimodal speech recognition, especially when acoustic data is heavily noised or even inaccessible. In this paper, we present a novel method for visual speech recognition. We benchmark it on the famous LRW lip-reading dataset by outperforming the existing approaches. After a comprehensive evaluation, we adapt the developed method and test it on the collected RUSAVIC corpus we recorded in-the-wild for vehicle driver. The results obtained demonstrate not only the high performance of the proposed method, but also the fundamental possibility of recognizing speech only by using video modality, even in such difficult natural conditions as driving.","url_abs":"https://eurasip.org/Proceedings/Eusipco/Eusipco2022/pdfs/0001131.pdf","url_pdf":"https://eurasip.org/Proceedings/Eusipco/Eusipco2022/pdfs/0001131.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"data-augmentation","task_name":"Data Augmentation"},{"task_slug":"lip-reading","task_name":"Lip Reading"},{"task_slug":"lipreading","task_name":"Lipreading"},{"task_slug":"speech-recognition","task_name":"Speech Recognition"},{"task_slug":"visual-speech-recognition","task_name":"Visual Speech Recognition"},{"task_slug":"speech-recognition-1","task_name":"speech-recognition"}],"methods":[{"method_slug":"label-smoothing","method_name":"Label Smoothing"},{"method_slug":"mixup","method_name":"Mixup"},{"method_slug":"test","method_name":"Test"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/lipreading-on-lip-reading-in-the-wild","task":"Lipreading","dataset":"Lip Reading in the Wild","model":"Vosk + MediaPipe + LS + MixUp + SA + 3DResNet-18 + BiLSTM + Cosine WR","rank_in_archive_order":6,"of":22,"metrics":{"Top-1 Accuracy":"88.7"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}