{"url":"/task/phoneme-recognition","name":"Phoneme Recognition","slug":"phoneme-recognition","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":104,"papers_with_code":27,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/timit","name":"TIMIT","full_name":"TIMIT Acoustic-Phonetic Continuous Speech Corpus","num_papers_in_archive":31}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":27,"of":27,"tagged_in_all":104,"items":[{"url":"/paper/wavenet-a-generative-model-for-raw-audio","title":"WaveNet: A Generative Model for Raw Audio","date":"2016-09-12","arxiv_id":"1609.03499","repositories_listed":62,"syntology":{"n":103,"n_ran":41,"n_unverified":62,"n_pointer_only":25}},{"url":"/paper/attention-based-models-for-speech-recognition","title":"Attention-Based Models for Speech Recognition","date":"2015-06-24","arxiv_id":"1506.07503","repositories_listed":14,"syntology":null},{"url":"/paper/sequence-transduction-with-recurrent-neural","title":"Sequence Transduction with Recurrent Neural Networks","date":"2012-11-14","arxiv_id":"1211.3711","repositories_listed":7,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/speech-recognition-with-deep-recurrent-neural","title":"Speech Recognition with Deep Recurrent Neural Networks","date":"2013-03-22","arxiv_id":"1303.5778","repositories_listed":5,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/improving-mispronunciation-detection-with","title":"Improving Mispronunciation Detection with Wav2vec2-based Momentum Pseudo-Labeling for Accentedness and Intelligibility Assessment","date":"2022-03-29","arxiv_id":"2203.15937","repositories_listed":2,"syntology":null},{"url":"/paper/simple-and-effective-zero-shot-cross-lingual","title":"Simple and Effective Zero-shot Cross-lingual Phoneme Recognition","date":"2021-09-23","arxiv_id":"2109.11680","repositories_listed":2,"syntology":null},{"url":"/paper/do-deep-nets-really-need-to-be-deep","title":"Do Deep Nets Really Need to be Deep?","date":"2013-12-21","arxiv_id":"1312.6184","repositories_listed":2,"syntology":null},{"url":"/paper/laser-learning-by-aligning-self-supervised","title":"LASER: Learning by Aligning Self-supervised Representations of Speech for Improving Content-related Tasks","date":"2024-06-13","arxiv_id":"2406.09153","repositories_listed":1,"syntology":null},{"url":"/paper/score-self-supervised-correspondence-fine","title":"SCORE: Self-supervised Correspondence Fine-tuning for Improved Content Representations","date":"2024-03-10","arxiv_id":"2403.06260","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-self-supervised-learning-models","title":"Fine-Tuning Self-Supervised Learning Models for End-to-End Pronunciation Scoring","date":"2023-09-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/allophant-cross-lingual-phoneme-recognition","title":"Allophant: Cross-lingual Phoneme Recognition with Articulatory Attributes","date":"2023-06-07","arxiv_id":"2306.04306","repositories_listed":1,"syntology":null},{"url":"/paper/mlphon-a-multifunctional-grapheme-phoneme","title":"Mlphon: A Multifunctional Grapheme-Phoneme Conversion Tool Using Finite State Transducers","date":"2022-09-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fithubert-going-thinner-and-deeper-for","title":"FitHuBERT: Going Thinner and Deeper for Knowledge Distillation of Speech Self-Supervised Learning","date":"2022-07-01","arxiv_id":"2207.00555","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/predicting-within-and-across-language-phoneme","title":"Predicting within and across language phoneme recognition performance of self-supervised learning speech pre-trained models","date":"2022-06-24","arxiv_id":"2206.12489","repositories_listed":1,"syntology":null},{"url":"/paper/text-aware-end-to-end-mispronunciation","title":"Text-Aware End-to-end Mispronunciation Detection and Diagnosis","date":"2022-06-15","arxiv_id":"2206.07289","repositories_listed":1,"syntology":null},{"url":"/paper/deep-neural-convolutive-matrix-factorization","title":"Deep Neural Convolutive Matrix Factorization for Articulatory Representation Decomposition","date":"2022-04-01","arxiv_id":"2204.00465","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-generative-latent-variable","title":"Benchmarking Generative Latent Variable Models for Speech","date":"2022-02-22","arxiv_id":"2202.12707","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/spanish-and-english-phoneme-recognition-by","title":"Spanish and English Phoneme Recognition by Training on Simulated Classroom Audio Recordings of Collaborative Learning Environments","date":"2022-02-21","arxiv_id":"2202.10536","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-semantic-driven-phoneme","title":"Self-supervised Semantic-driven Phoneme Discovery for Zero-resource Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/singing-language-identification-using-a-deep","title":"Singing Language Identification using a Deep Phonotactic Approach","date":"2021-05-31","arxiv_id":"2105.15014","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-low-resource-phoneme-recognition-on","title":"Real-time low-resource phoneme recognition on edge devices","date":"2021-03-25","arxiv_id":"2103.13997","repositories_listed":1,"syntology":null},{"url":"/paper/word-error-rate-estimation-without-asr-output","title":"Word Error Rate Estimation Without ASR Output: e-WER2","date":"2020-08-08","arxiv_id":"2008.03403","repositories_listed":1,"syntology":null},{"url":"/paper/application-of-word2vec-in-phoneme","title":"Application of Word2vec in Phoneme Recognition","date":"2019-12-17","arxiv_id":"1912.08011","repositories_listed":1,"syntology":null},{"url":"/paper/quaternion-convolutional-neural-networks-for-1","title":"Quaternion Convolutional Neural Networks for End-to-End Automatic Speech Recognition","date":"2018-06-20","arxiv_id":"1806.07789","repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-speech-recognition-with","title":"Towards End-to-End Speech Recognition with Deep Convolutional Neural Networks","date":"2017-01-10","arxiv_id":"1701.02720","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/regularizing-rnns-by-stabilizing-activations","title":"Regularizing RNNs by Stabilizing Activations","date":"2015-11-26","arxiv_id":"1511.08400","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-phoneme-sequence-recognition-using","title":"End-to-end Phoneme Sequence Recognition using Convolutional Neural Networks","date":"2013-12-07","arxiv_id":"1312.2137","repositories_listed":1,"syntology":null}],"syntology_records":6,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}