{"url":"/task/speech-representation-learning","name":"Speech Representation Learning","slug":"speech-representation-learning","description_markdown":null,"categories":[{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":131,"papers_with_code":48,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":48,"tagged_in_all":131,"items":[{"url":"/paper/hubert-self-supervised-speech-representation","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","date":"2021-06-14","arxiv_id":"2106.07447","repositories_listed":11,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/mockingjay-unsupervised-speech-representation","title":"Mockingjay: Unsupervised Speech Representation Learning with Deep Bidirectional Transformer Encoders","date":"2019-10-25","arxiv_id":"1910.12638","repositories_listed":7,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/unispeech-sat-universal-speech-representation","title":"UniSpeech-SAT: Universal Speech Representation Learning with Speaker Aware Pre-Training","date":"2021-10-12","arxiv_id":"2110.05752","repositories_listed":5,"syntology":null},{"url":"/paper/unispeech-unified-speech-representation","title":"UniSpeech: Unified Speech Representation Learning with Labeled and Unlabeled Data","date":"2021-01-19","arxiv_id":"2101.07597","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/unsupervised-speech-representation-learning","title":"Unsupervised speech representation learning using WaveNet autoencoders","date":"2019-01-25","arxiv_id":"1901.08810","repositories_listed":5,"syntology":null},{"url":"/paper/w2v-bert-combining-contrastive-learning-and","title":"W2v-BERT: Combining Contrastive Learning and Masked Language Modeling for Self-Supervised Speech Pre-Training","date":"2021-08-07","arxiv_id":"2108.06209","repositories_listed":4,"syntology":null},{"url":"/paper/an-unsupervised-autoregressive-model-for","title":"An Unsupervised Autoregressive Model for Speech Representation Learning","date":"2019-04-05","arxiv_id":"1904.03240","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/ernie-sat-speech-and-text-joint-pretraining-1","title":"ERNIE-SAT: Speech and Text Joint Pretraining for Cross-Lingual Multi-Speaker Text-to-Speech","date":"2022-11-07","arxiv_id":"2211.03545","repositories_listed":2,"syntology":null},{"url":"/paper/a-3-t-alignment-aware-acoustic-and-text","title":"A$^3$T: Alignment-Aware Acoustic and Text Pretraining for Speech Synthesis and Editing","date":"2022-03-18","arxiv_id":"2203.09690","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/learning-audio-visual-speech-representation-1","title":"Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction","date":"2022-01-05","arxiv_id":"2201.02184","repositories_listed":2,"syntology":null},{"url":"/paper/xls-r-self-supervised-cross-lingual-speech","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","date":"2021-11-17","arxiv_id":"2111.09296","repositories_listed":2,"syntology":null},{"url":"/paper/sampling-strategies-in-siamese-networks-for","title":"Sampling strategies in Siamese Networks for unsupervised speech representation learning","date":"2018-04-30","arxiv_id":"1804.11297","repositories_listed":2,"syntology":null},{"url":"/paper/multi-task-corrupted-prediction-for-learning","title":"Multi-Task Corrupted Prediction for Learning Robust Audio-Visual Speech Representation","date":"2025-01-23","arxiv_id":"2504.18539","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/k2ssl-a-faster-and-better-framework-for-self","title":"k2SSL: A Faster and Better Framework for Self-Supervised Speech Representation Learning","date":"2024-11-26","arxiv_id":"2411.17100","repositories_listed":1,"syntology":null},{"url":"/paper/eh-mam-easy-to-hard-masked-acoustic-modeling","title":"EH-MAM: Easy-to-Hard Masked Acoustic Modeling for Self-Supervised Speech Representation Learning","date":"2024-10-17","arxiv_id":"2410.13179","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/self-supervised-syllable-discovery-based-on","title":"Self-Supervised Syllable Discovery Based on Speaker-Disentangled HuBERT","date":"2024-09-16","arxiv_id":"2409.10103","repositories_listed":1,"syntology":null},{"url":"/paper/mhubert-147-a-compact-multilingual-hubert","title":"mHuBERT-147: A Compact Multilingual HuBERT Model","date":"2024-06-10","arxiv_id":"2406.06371","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-end-to-end-approach-to-noise","title":"An Efficient End-to-End Approach to Noise Invariant Speech Features via Multi-Task Learning","date":"2024-03-13","arxiv_id":"2403.08654","repositories_listed":1,"syntology":null},{"url":"/paper/the-effect-of-batch-size-on-contrastive-self","title":"The Effect of Batch Size on Contrastive Self-Supervised Speech Representation Learning","date":"2024-02-21","arxiv_id":"2402.13723","repositories_listed":1,"syntology":null},{"url":"/paper/clara-multilingual-contrastive-learning-for","title":"CLARA: Multilingual Contrastive Learning for Audio Representation Acquisition","date":"2023-10-18","arxiv_id":"2310.11830","repositories_listed":1,"syntology":null},{"url":"/paper/must-p-srl-multi-lingual-and-unified","title":"MUST&P-SRL: Multi-lingual and Unified Syllabification in Text and Phonetic Domains for Speech Representation Learning","date":"2023-10-17","arxiv_id":"2310.11541","repositories_listed":1,"syntology":null},{"url":"/paper/fast-hubert-an-efficient-training-framework","title":"Fast-HuBERT: An Efficient Training Framework for Self-Supervised Speech Representation Learning","date":"2023-09-25","arxiv_id":"2309.13860","repositories_listed":1,"syntology":null},{"url":"/paper/qs-tts-towards-semi-supervised-text-to-speech","title":"QS-TTS: Towards Semi-Supervised Text-to-Speech Synthesis via Vector-Quantized Self-Supervised Speech Representation Learning","date":"2023-08-31","arxiv_id":"2309.00126","repositories_listed":1,"syntology":null},{"url":"/paper/dinosr-self-distillation-and-online","title":"DinoSR: Self-Distillation and Online Clustering for Self-supervised Speech Representation Learning","date":"2023-05-17","arxiv_id":"2305.10005","repositories_listed":1,"syntology":null},{"url":"/paper/a-multimodal-dynamical-variational","title":"A multimodal dynamical variational autoencoder for audiovisual speech representation learning","date":"2023-05-05","arxiv_id":"2305.03582","repositories_listed":1,"syntology":null},{"url":"/paper/facexhubert-text-less-speech-driven-e-x","title":"FaceXHuBERT: Text-less Speech-driven E(X)pressive 3D Facial Animation Synthesis Using Self-Supervised Speech Representation Learning","date":"2023-03-09","arxiv_id":"2303.05416","repositories_listed":1,"syntology":null},{"url":"/paper/low-latency-transformers-for-speech","title":"A low latency attention module for streaming self-supervised speech representation learning","date":"2023-02-27","arxiv_id":"2302.13451","repositories_listed":1,"syntology":null},{"url":"/paper/structured-pruning-of-self-supervised-pre","title":"Structured Pruning of Self-Supervised Pre-trained Models for Speech Recognition and Understanding","date":"2023-02-27","arxiv_id":"2302.14132","repositories_listed":1,"syntology":null},{"url":"/paper/mt4ssl-boosting-self-supervised-speech","title":"MT4SSL: Boosting Self-Supervised Speech Representation Learning by Integrating Multiple Targets","date":"2022-11-14","arxiv_id":"2211.07321","repositories_listed":1,"syntology":null},{"url":"/paper/data2vec-aqc-search-for-the-right-teaching","title":"data2vec-aqc: Search for the right Teaching Assistant in the Teacher-Student training setup","date":"2022-11-02","arxiv_id":"2211.01246","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}