{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/lip-reading/papers/2","list_of":"/task/lip-reading","task":"Lip Reading","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,153],"of":153,"counts":{"archive_papers_tagged":153,"with_a_code_link":49,"where_syntology_ran_a_sample":17,"not_listed_spam_title":0,"listed":153,"listed_where_code_ran":17,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":14,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":14,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/lip-reading","prev":"/task/lip-reading","next":null,"papers":[{"url":null,"slug":"leveraging-uni-modal-self-supervised-learning","title":"Leveraging Uni-Modal Self-Supervised Learning for Multimodal Audio-visual Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-and-challenges-in-deep-lip-reading","title":"Advances and Challenges in Deep Lip Reading","date":"2021-10-15","arxiv_id":"2110.07879","repositories_listed":0,"syntology":null},{"url":"/paper/sub-word-level-lip-reading-with-visual","slug":"sub-word-level-lip-reading-with-visual","title":"Sub-word Level Lip Reading With Visual Attention","date":"2021-10-14","arxiv_id":"2110.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-point-identifying-critical","title":"Perception Point: Identifying Critical Learning Periods in Speech for Bilingual Networks","date":"2021-10-13","arxiv_id":"2110.06507","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-recognition-is-worth-32","title":"Audio-Visual Speech Recognition is Worth 32$\\times$32$\\times$8 Voxels","date":"2021-09-20","arxiv_id":"2109.09536","repositories_listed":0,"syntology":null},{"url":null,"slug":"lrwr-large-scale-benchmark-for-lip-reading-in","title":"LRWR: Large-Scale Benchmark for Lip Reading in Russian language","date":"2021-09-14","arxiv_id":"2109.06692","repositories_listed":0,"syntology":null},{"url":null,"slug":"simullr-simultaneous-lip-reading-transducer","title":"SimulLR: Simultaneous Lip Reading Transducer with Attention-Guided Adaptive Memory","date":"2021-08-31","arxiv_id":"2108.13630","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-semantic-spatio-temporal-graph","slug":"adaptive-semantic-spatio-temporal-graph","title":"Adaptive Semantic-Spatio-Temporal Graph Convolutional Network for Lip Reading","date":"2021-08-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-attention-mechanism-and","title":"Spatio-Temporal Attention Mechanism and Knowledge Distillation for Lip Reading","date":"2021-08-07","arxiv_id":"2108.03543","repositories_listed":0,"syntology":null},{"url":null,"slug":"facetron-multi-speaker-face-to-speech-model","title":"Facetron: A Multi-speaker Face-to-Speech Model based on Cross-modal Latent Representations","date":"2021-07-26","arxiv_id":"2107.12003","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-the-master-distilling-cross","title":"Learning From the Master: Distilling Cross-Modal Advanced Knowledge for Lip Reading","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lira-learning-visual-speech-representations","title":"LiRA: Learning Visual Speech Representations from Audio through Self-supervision","date":"2021-06-16","arxiv_id":"2106.09171","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-perspective-lstm-for-joint-visual","title":"Multi-Perspective LSTM for Joint Visual Representation Learning","date":"2021-05-06","arxiv_id":"2105.02802","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-video-to-speech-synthesis-using","title":"End-to-End Video-To-Speech Synthesis using Generative Adversarial Networks","date":"2021-04-27","arxiv_id":"2104.13332","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-information-streams-in-end-to-end","title":"Fusing information streams in end-to-end audio-visual speech recognition","date":"2021-04-19","arxiv_id":"2104.09482","repositories_listed":0,"syntology":null},{"url":null,"slug":"lip-reading-using-external-viseme-decoding","title":"Lip reading using external viseme decoding","date":"2021-04-10","arxiv_id":"2104.04784","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-self-supervised-learning-of","title":"Contrastive Self-Supervised Learning of Global-Local Audio-Visual Representations","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lip-reading-with-hierarchical-pyramidal","title":"Lip-reading with Hierarchical Pyramidal Convolution and Self-Attention","date":"2020-12-28","arxiv_id":"2012.14360","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-homophemes-in-lip-reading-using","title":"Disentangling Homophemes in Lip Reading using Perplexity Analysis","date":"2020-11-28","arxiv_id":"2012.07528","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-lip-localization-techniques-used","title":"A Study on Lip Localization Techniques used for Lip reading from a Video","date":"2020-09-28","arxiv_id":"2009.13420","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-voices-and-hearing-voices-learning","title":"Seeing voices and hearing voices: learning discriminative embeddings using cross-modal self-supervision","date":"2020-04-29","arxiv_id":"2004.14326","repositories_listed":0,"syntology":null},{"url":"/paper/pseudo-convolutional-policy-gradient-for","slug":"pseudo-convolutional-policy-gradient-for","title":"Pseudo-Convolutional Policy Gradient for Sequence-to-Sequence Lip-Reading","date":"2020-03-09","arxiv_id":"2003.03983","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-synchronization-using-the-hand-preceding","title":"Re-synchronization using the Hand Preceding Model for Multi-modal Fusion in Automatic Continuous Cued Speech Recognition","date":"2020-02-23","arxiv_id":"2001.00854","repositories_listed":0,"syntology":null},{"url":"/paper/asr-is-all-you-need-cross-modal-distillation","slug":"asr-is-all-you-need-cross-modal-distillation","title":"ASR is all you need: cross-modal distillation for lip reading","date":"2019-11-28","arxiv_id":"1911.12747","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-pose-invariant-lip-reading","title":"Towards Pose-invariant Lip-Reading","date":"2019-11-14","arxiv_id":"1911.06095","repositories_listed":0,"syntology":null},{"url":"/paper/spatio-temporal-fusion-based-convolutional","slug":"spatio-temporal-fusion-based-convolutional","title":"Spatio-Temporal Fusion Based Convolutional Sequence Learning for Lip Reading","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/multi-grained-spatio-temporal-modeling-for","slug":"multi-grained-spatio-temporal-modeling-for","title":"Multi-Grained Spatio-temporal Modeling for Lip-reading","date":"2019-08-30","arxiv_id":"1908.11618","repositories_listed":0,"syntology":null},{"url":"/paper/a-cascade-sequence-to-sequence-model-for","slug":"a-cascade-sequence-to-sequence-model-for","title":"A Cascade Sequence-to-Sequence Model for Chinese Mandarin Lip Reading","date":"2019-08-14","arxiv_id":"1908.04917","repositories_listed":0,"syntology":null},{"url":null,"slug":"realistic-speech-driven-facial-animation-with","title":"Realistic Speech-Driven Facial Animation with GANs","date":"2019-06-14","arxiv_id":"1906.06337","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobivsr-a-visual-speech-recognition-solution","title":"MobiVSR: A Visual Speech Recognition Solution for Mobile Devices","date":"2019-05-10","arxiv_id":"1905.03968","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesising-3d-facial-motion-from-in-the","title":"Synthesising 3D Facial Motion from \"In-the-Wild\" Speech","date":"2019-04-15","arxiv_id":"1904.07002","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-videos-with-deep-convolutional","title":"Learning from Videos with Deep Convolutional LSTM Networks","date":"2019-04-09","arxiv_id":"1904.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-analysis-of-deep-audio-visual","title":"An Empirical Analysis of Deep Audio-Visual Models for Speech Recognition","date":"2018-12-21","arxiv_id":"1812.09336","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-audio-visual-switching-for-speech","title":"Contextual Audio-Visual Switching For Speech Enhancement in Real-World Environments","date":"2018-08-28","arxiv_id":"1808.09825","repositories_listed":0,"syntology":null},{"url":null,"slug":"lip-reading-driven-deep-learning-approach-for","title":"Lip-Reading Driven Deep Learning Approach for Speech Enhancement","date":"2018-07-31","arxiv_id":"1808.00046","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-lip-reading-a-comparison-of-models-and","title":"Deep Lip Reading: a comparison of models and an online application","date":"2018-06-15","arxiv_id":"1806.06053","repositories_listed":0,"syntology":null},{"url":null,"slug":"lip-reading-using-convolutional-auto-encoders","title":"Lip Reading Using Convolutional Auto Encoders as Feature Extractor","date":"2018-05-31","arxiv_id":"1805.12371","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-aware-design-of-a-deep-convolutional","title":"Resource aware design of a deep convolutional-recurrent neural network for speech recognition through audio-visual sensor fusion","date":"2018-03-13","arxiv_id":"1803.04840","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-lip-reading-using-audio","title":"Deep Learning for Lip Reading using Audio-Visual Information for Urdu Language","date":"2018-02-15","arxiv_id":"1802.05521","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-visemes-improving-machine-lipreading-1","title":"Decoding visemes: improving machine lipreading","date":"2017-10-03","arxiv_id":"1710.01169","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-phonemes-improving-machine-lip","title":"Finding phonemes: improving machine lip-reading","date":"2017-10-03","arxiv_id":"1710.01142","repositories_listed":0,"syntology":null},{"url":null,"slug":"resolution-limits-on-visual-speech","title":"Resolution limits on visual speech recognition","date":"2017-10-03","arxiv_id":"1710.01073","repositories_listed":0,"syntology":null},{"url":null,"slug":"some-observations-on-computer-lip-reading","title":"Some observations on computer lip-reading: moving from the dream to the reality","date":"2017-10-03","arxiv_id":"1710.01084","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-independent-machine-lip-reading-with","title":"Speaker-independent machine lip-reading with speaker-dependent viseme classifiers","date":"2017-10-03","arxiv_id":"1710.01122","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-phoneme-to-viseme-maps-best-improve","title":"Which phoneme-to-viseme maps best improve visual-only computer lip-reading?","date":"2017-10-03","arxiv_id":"1710.01093","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-viseme-vocabulary-construction-to","title":"Automatic Viseme Vocabulary Construction to Enhance Continuous Lip-reading","date":"2017-04-26","arxiv_id":"1704.08035","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-estimating-the-upper-bound-of-visual","title":"Towards Estimating the Upper Bound of Visual-Speech Recognition: The Visual Lip-Reading Feasibility Database","date":"2017-04-26","arxiv_id":"1704.08028","repositories_listed":0,"syntology":null},{"url":"/paper/lip-reading-sentences-in-the-wild","slug":"lip-reading-sentences-in-the-wild","title":"Lip Reading Sentences in the Wild","date":"2016-11-16","arxiv_id":"1611.05358","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-based-action-recognition-using-rate","title":"Video-Based Action Recognition Using Rate-Invariant Analysis of Covariance Trajectories","date":"2015-03-23","arxiv_id":"1503.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"definition-of-visual-speech-element-and","title":"Definition of Visual Speech Element and Research on a Method of Extracting Feature Vector for Korean Lip-Reading","date":"2014-11-15","arxiv_id":"1411.4114","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-words-for-automatic-lip-reading","title":"Visual Words for Automatic Lip-Reading","date":"2014-09-17","arxiv_id":"1409.6689","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-speech-recognition","title":"Visual Speech Recognition","date":"2014-09-03","arxiv_id":"1409.1411","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-passwords-using-automatic-lip-reading","title":"Visual Passwords Using Automatic Lip Reading","date":"2014-09-02","arxiv_id":"1409.0924","repositories_listed":0,"syntology":null}],"record_sha256":"0d5cd49c109192e895305a109108e4cfa842af1f0ec116eb76dd6e478deb744a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}