{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-speech-recognition/papers/2","list_of":"/task/visual-speech-recognition","task":"Visual Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,182],"of":182,"counts":{"archive_papers_tagged":182,"with_a_code_link":62,"where_syntology_ran_a_sample":19,"not_listed_spam_title":0,"listed":182,"listed_where_code_ran":19,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":16,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":16,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-speech-recognition","prev":"/task/visual-speech-recognition","next":null,"papers":[{"url":"/paper/lip2vec-efficient-and-robust-visual-speech","slug":"lip2vec-efficient-and-robust-visual-speech","title":"Lip2Vec: Efficient and Robust Visual Speech Recognition via Latent-to-Latent Visual to Audio Representation Mapping","date":"2023-08-11","arxiv_id":"2308.06112","repositories_listed":0,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":16,"n_pointer_only":19,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 1 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lip2vec-efficient-and-robust-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2308.06112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06112"}},"official":null}},{"url":null,"slug":"sparsevsr-lightweight-and-noise-robust-visual","title":"SparseVSR: Lightweight and Noise Robust Visual Speech Recognition","date":"2023-07-10","arxiv_id":"2307.04552","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-speaker-independent-visual-speech","title":"Automated Speaker Independent Visual Speech Recognition: A Comprehensive Survey","date":"2023-06-14","arxiv_id":"2306.08314","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-gap-in-visual-speech","title":"Improving the Gap in Visual Speech Recognition Between Normal and Silent Speech Based on Metric Learning","date":"2023-05-23","arxiv_id":"2305.14203","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-temporal-lip-audio-memory-for-visual","title":"Multi-Temporal Lip-Audio Memory for Visual Speech Recognition","date":"2023-05-08","arxiv_id":"2305.04542","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-spatio-temporal-facial","title":"Deep Learning-based Spatio Temporal Facial Feature Visual Speech Recognition","date":"2023-04-30","arxiv_id":"2305.00552","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthvsr-scaling-up-visual-speech-recognition","title":"SynthVSR: Scaling Up Visual Speech Recognition With Synthetic Supervision","date":"2023-03-30","arxiv_id":"2303.17200","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-npu-aslp-system-for-audio-visual-speech","title":"The NPU-ASLP System for Audio-Visual Speech Recognition in MISP 2022 Challenge","date":"2023-03-11","arxiv_id":"2303.06341","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-visual-forced-alignment-learning-to","title":"Deep Visual Forced Alignment: Learning to Align Transcription with Talking Face Video","date":"2023-02-27","arxiv_id":"2303.08670","repositories_listed":0,"syntology":null},{"url":"/paper/audio-visual-speech-and-gesture-recognition","slug":"audio-visual-speech-and-gesture-recognition","title":"Audio-Visual Speech and Gesture Recognition by Sensors of Mobile Devices","date":"2023-02-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/conformers-are-all-you-need-for-visual-speech","slug":"conformers-are-all-you-need-for-visual-speech","title":"Conformers are All You Need for Visual Speech Recognition","date":"2023-02-17","arxiv_id":"2302.10915","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-tuning-of-deep-neural-networks-for","title":"Prompt Tuning of Deep Neural Networks for Speaker-adaptive Visual Speech Recognition","date":"2023-02-16","arxiv_id":"2302.08102","repositories_listed":0,"syntology":null},{"url":null,"slug":"av-data2vec-self-supervised-learning-of-audio","title":"AV-data2vec: Self-supervised Learning of Audio-Visual Speech Representations with Contextualized Target Representations","date":"2023-02-10","arxiv_id":"2302.06419","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-purpose-audio-visual-corpus-for-multi","title":"A Multi-Purpose Audio-Visual Corpus for Multi-Modal Persian Speech Recognition: the Arman-AV Dataset","date":"2023-01-21","arxiv_id":"2301.10180","repositories_listed":0,"syntology":null},{"url":null,"slug":"revise-self-supervised-speech-resynthesis-1","title":"ReVISE: Self-Supervised Speech Resynthesis With Visual Input for Universal and Generalized Speech Regeneration","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/revise-self-supervised-speech-resynthesis","slug":"revise-self-supervised-speech-resynthesis","title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","date":"2022-12-21","arxiv_id":"2212.11377","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-modality-specific-representations","title":"Leveraging Modality-specific Representations for Audio-visual Speech Recognition via Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05301","repositories_listed":0,"syntology":null},{"url":null,"slug":"vatlm-visual-audio-text-pre-training-with","title":"VATLM: Visual-Audio-Text Pre-Training with Unified Masked Prediction for Speech Representation Learning","date":"2022-11-21","arxiv_id":"2211.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-audio-visual-speech-recognition","title":"Streaming Audio-Visual Speech Recognition with Alignment Regularization","date":"2022-11-03","arxiv_id":"2211.02133","repositories_listed":0,"syntology":null},{"url":"/paper/visual-speech-recognition-in-a-driver","slug":"visual-speech-recognition-in-a-driver","title":"Visual Speech Recognition in a Driver Assistance System","date":"2022-08-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kaggle-competition-cantonese-audio-visual","title":"Kaggle Competition: Cantonese Audio-Visual Speech Recognition for In-car Commands","date":"2022-07-06","arxiv_id":"2207.02663","repositories_listed":0,"syntology":null},{"url":null,"slug":"lip-listening-mixing-senses-to-understand","title":"Lip-Listening: Mixing Senses to Understand Lips using Cross Modality Knowledge Distillation for Word-Based Models","date":"2022-06-05","arxiv_id":"2207.05692","repositories_listed":0,"syntology":null},{"url":null,"slug":"rusavic-corpus-russian-audio-visual-speech-in","title":"RUSAVIC Corpus: Russian Audio-Visual Speech in Cars","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"is-lip-region-of-interest-sufficient-for","title":"Is Lip Region-of-Interest Sufficient for Lipreading?","date":"2022-05-28","arxiv_id":"2205.14295","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-visual-speech-analysis-a","title":"Deep Learning for Visual Speech Analysis: A Survey","date":"2022-05-22","arxiv_id":"2205.10839","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-contextually-fused-audio-visual","title":"Learning Contextually Fused Audio-visual Representations for Audio-visual Speech Recognition","date":"2022-02-15","arxiv_id":"2202.07428","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-video-front-ends-for-audio","title":"Transformer-Based Video Front-Ends for Audio-Visual Speech Recognition for Single and Multi-Person Video","date":"2022-01-25","arxiv_id":"2201.10439","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-progress-in-the-cuhk-dysarthric-speech","title":"Recent Progress in the CUHK Dysarthric Speech Recognition System","date":"2022-01-15","arxiv_id":"2201.05845","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-uni-modal-self-supervised-learning","title":"Leveraging Uni-Modal Self-Supervised Learning for Multimodal Audio-visual Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-and-challenges-in-deep-lip-reading","title":"Advances and Challenges in Deep Lip Reading","date":"2021-10-15","arxiv_id":"2110.07879","repositories_listed":0,"syntology":null},{"url":"/paper/sub-word-level-lip-reading-with-visual","slug":"sub-word-level-lip-reading-with-visual","title":"Sub-word Level Lip Reading With Visual Attention","date":"2021-10-14","arxiv_id":"2110.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-point-identifying-critical","title":"Perception Point: Identifying Critical Learning Periods in Speech for Bilingual Networks","date":"2021-10-13","arxiv_id":"2110.06507","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-recognition-is-worth-32","title":"Audio-Visual Speech Recognition is Worth 32$\\times$32$\\times$8 Voxels","date":"2021-09-20","arxiv_id":"2109.09536","repositories_listed":0,"syntology":null},{"url":null,"slug":"lrwr-large-scale-benchmark-for-lip-reading-in","title":"LRWR: Large-Scale Benchmark for Lip Reading in Russian language","date":"2021-09-14","arxiv_id":"2109.06692","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-vocabulary-audio-visual-speech","title":"Large-vocabulary Audio-visual Speech Recognition in Noisy Environments","date":"2021-09-10","arxiv_id":"2109.04894","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-attention-mechanism-and","title":"Spatio-Temporal Attention Mechanism and Knowledge Distillation for Lip Reading","date":"2021-08-07","arxiv_id":"2108.03543","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-decoding-of-words-from-visual","title":"Interactive decoding of words from visual speech recognition models","date":"2021-07-01","arxiv_id":"2107.00692","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-information-streams-in-end-to-end","title":"Fusing information streams in end-to-end audio-visual speech recognition","date":"2021-04-19","arxiv_id":"2104.09482","repositories_listed":0,"syntology":null},{"url":"/paper/part-based-lipreading-for-audio-visual-speech","slug":"part-based-lipreading-for-audio-visual-speech","title":"Part-based Lipreading for Audio-Visual Speech Recognition","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/lip-graph-assisted-audio-visual-speech","slug":"lip-graph-assisted-audio-visual-speech","title":"Lip Graph Assisted Audio-Visual Speech Recognition Using Bidirectional Synchronous Fusion","date":"2020-10-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"notic-my-speech-blending-speech-patterns-with","title":"\"Notic My Speech\" -- Blending Speech Patterns With Multimedia","date":"2020-06-12","arxiv_id":"2006.08599","repositories_listed":0,"syntology":null},{"url":"/paper/audio-visual-recognition-of-overlapped-speech","slug":"audio-visual-recognition-of-overlapped-speech","title":"Audio-visual Recognition of Overlapped speech for the LRS2 dataset","date":"2020-01-06","arxiv_id":"2001.01656","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-adversarial-attacks-on-audio-visual","title":"Detecting Adversarial Attacks On Audiovisual Speech Recognition","date":"2019-12-18","arxiv_id":"1912.08639","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-speech-recognition-using-eeg-and","title":"Continuous Speech Recognition using EEG and Video","date":"2019-12-16","arxiv_id":"1912.07730","repositories_listed":0,"syntology":null},{"url":"/paper/asr-is-all-you-need-cross-modal-distillation","slug":"asr-is-all-you-need-cross-modal-distillation","title":"ASR is all you need: cross-modal distillation for lip reading","date":"2019-11-28","arxiv_id":"1911.12747","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-lombard-effect-influence-on","title":"Investigating the Lombard Effect Influence on End-to-End Audio-Visual Speech Recognition","date":"2019-06-05","arxiv_id":"1906.02112","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobivsr-a-visual-speech-recognition-solution","title":"MobiVSR: A Visual Speech Recognition Solution for Mobile Devices","date":"2019-05-10","arxiv_id":"1905.03968","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-visual-speech-recognition-for","title":"End-to-End Visual Speech Recognition for Small-Scale Datasets","date":"2019-04-02","arxiv_id":"1904.01954","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-attention-for-end-to-end-audio","title":"Modality Attention for End-to-End Audio-visual Speech Recognition","date":"2018-11-13","arxiv_id":"1811.05250","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-feature-pyramid-attention-module-for","title":"3D Feature Pyramid Attention Module for Robust Visual Speech Recognition","date":"2018-10-15","arxiv_id":"1810.06178","repositories_listed":0,"syntology":null},{"url":"/paper/audio-visual-speech-recognition-with-a-hybrid","slug":"audio-visual-speech-recognition-with-a-hybrid","title":"Audio-Visual Speech Recognition With A Hybrid CTC/Attention Architecture","date":"2018-09-28","arxiv_id":"1810.00108","repositories_listed":0,"syntology":null},{"url":null,"slug":"perfect-match-improved-cross-modal-embeddings","title":"Perfect match: Improved cross-modal embeddings for audio-visual synchronisation","date":"2018-09-21","arxiv_id":"1809.08001","repositories_listed":0,"syntology":null},{"url":"/paper/large-scale-visual-speech-recognition","slug":"large-scale-visual-speech-recognition","title":"Large-Scale Visual Speech Recognition","date":"2018-07-13","arxiv_id":"1807.05162","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-lip-reading-a-comparison-of-models-and","title":"Deep Lip Reading: a comparison of models and an online application","date":"2018-06-15","arxiv_id":"1806.06053","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-lipreading-sentences-with-active","title":"Towards Lipreading Sentences with Active Appearance Models","date":"2018-05-29","arxiv_id":"1805.11688","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-dependent-modulation-of-the-visual","title":"Task-dependent modulation of the visual sensory thalamus assists visual-speech recognition","date":"2018-05-24","arxiv_id":"1805.05682","repositories_listed":0,"syntology":null},{"url":"/paper/visual-only-recognition-of-normal-whispered","slug":"visual-only-recognition-of-normal-whispered","title":"Visual-Only Recognition of Normal, Whispered and Silent Speech","date":"2018-02-18","arxiv_id":"1802.06399","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-multiple-views-for-visual-speech","title":"Combining Multiple Views for Visual Speech Recognition","date":"2017-10-19","arxiv_id":"1710.07168","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-speech-recognition-using-pca-networks","title":"Visual Speech Recognition Using PCA Networks and LSTMs in a Tandem GMM-HMM System","date":"2017-10-19","arxiv_id":"1710.07161","repositories_listed":0,"syntology":null},{"url":null,"slug":"resolution-limits-on-visual-speech","title":"Resolution limits on visual speech recognition","date":"2017-10-03","arxiv_id":"1710.01073","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-speech-recognition-aligning","title":"Visual speech recognition: aligning terminologies for better understanding","date":"2017-10-03","arxiv_id":"1710.01292","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-phoneme-to-viseme-maps-best-improve","title":"Which phoneme-to-viseme maps best improve visual-only computer lip-reading?","date":"2017-10-03","arxiv_id":"1710.01093","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-machine-learning-integrating","title":"Multimodal Machine Learning: Integrating Language, Vision and Speech","date":"2017-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-estimating-the-upper-bound-of-visual","title":"Towards Estimating the Upper Bound of Visual-Speech Recognition: The Visual Lip-Reading Feasibility Database","date":"2017-04-26","arxiv_id":"1704.08028","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multimodal-representation-learning-from","title":"Deep Multimodal Representation Learning from Temporal Data","date":"2017-04-11","arxiv_id":"1704.03152","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-visual-speech-recognition-with","title":"End-To-End Visual Speech Recognition With LSTMs","date":"2017-01-20","arxiv_id":"1701.05847","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-multimodal-lstm-for-audio-visual","title":"Auxiliary Multimodal LSTM for Audio-visual Speech Recognition and Lipreading","date":"2017-01-16","arxiv_id":"1701.04224","repositories_listed":0,"syntology":null},{"url":"/paper/lip-reading-sentences-in-the-wild","slug":"lip-reading-sentences-in-the-wild","title":"Lip Reading Sentences in the Wild","date":"2016-11-16","arxiv_id":"1611.05358","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-recognition-using-deep","title":"Audio Visual Speech Recognition using Deep Recurrent Neural Networks","date":"2016-11-09","arxiv_id":"1611.02879","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-three-dimensional-approach-to-visual-speech","title":"A three-dimensional approach to Visual Speech Recognition using Discrete Cosine Transforms","date":"2016-09-07","arxiv_id":"1609.01932","repositories_listed":0,"syntology":null},{"url":null,"slug":"manifold-kernels-comparison-in-mkpls-for","title":"Manifold-Kernels Comparison in MKPLS for Visual Speech Recognition","date":"2016-01-22","arxiv_id":"1601.05861","repositories_listed":0,"syntology":null},{"url":null,"slug":"listening-with-your-eyes-towards-a-practical","title":"Listening With Your Eyes: Towards a Practical Visual Speech Recognition System Using Deep Boltzmann Machines","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-based-action-recognition-using-rate","title":"Video-Based Action Recognition Using Rate-Invariant Analysis of Covariance Trajectories","date":"2015-03-23","arxiv_id":"1503.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multimodal-learning-for-audio-visual","title":"Deep Multimodal Learning for Audio-Visual Speech Recognition","date":"2015-01-22","arxiv_id":"1501.05396","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-words-for-automatic-lip-reading","title":"Visual Words for Automatic Lip-Reading","date":"2014-09-17","arxiv_id":"1409.6689","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-speech-recognition","title":"Visual Speech Recognition","date":"2014-09-03","arxiv_id":"1409.1411","repositories_listed":0,"syntology":null},{"url":null,"slug":"recognition-of-isolated-words-using-zernike","title":"Recognition of Isolated Words using Zernike and MFCC features for Audio Visual Speech Recognition","date":"2014-07-04","arxiv_id":"1407.1165","repositories_listed":0,"syntology":null},{"url":null,"slug":"preliminary-test-of-a-real-time-interactive","title":"Preliminary Test of a Real-Time, Interactive Silent Speech Interface Based on Electromagnetic Articulograph","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rate-invariant-analysis-of-trajectories-on","title":"Rate-Invariant Analysis of Trajectories on Riemannian Manifolds with Application in Visual Speech Recognition","date":"2014-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mkpls-manifold-kernel-partial-least-squares","title":"MKPLS: Manifold Kernel Partial Least Squares for Lipreading and Speaker Identification","date":"2013-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-synchronous-corpus-of-acoustic-and","title":"Building a synchronous corpus of acoustic and 3D facial marker data for adaptive audio-visual speech synthesis","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sutav-a-turkish-audio-visual-database","title":"SUTAV: A Turkish Audio-Visual Database","date":"2012-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"393f1cd8bd9d18b1d910cf8e2939dc2f36374bac2a2248828fd54e78c113126f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}