{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/wav2vec-2-0-a-framework-for-self-supervised","title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations","arxiv_id":"2006.11477","date":"2020-06-20","proceeding":"NeurIPS 2020 12","authors":["Alexei Baevski","Henry Zhou","Abdel-rahman Mohamed","Michael Auli"],"abstract":"We show for the first time that learning powerful representations from speech audio alone followed by fine-tuning on transcribed speech can outperform the best semi-supervised methods while being conceptually simpler. wav2vec 2.0 masks the speech input in the latent space and solves a contrastive task defined over a quantization of the latent representations which are jointly learned. Experiments using all labeled data of Librispeech achieve 1.8/3.3 WER on the clean/other test sets. When lowering the amount of labeled data to one hour, wav2vec 2.0 outperforms the previous state of the art on the 100 hour subset while using 100 times less labeled data. Using just ten minutes of labeled data and pre-training on 53k hours of unlabeled data still achieves 4.8/8.2 WER. This demonstrates the feasibility of speech recognition with limited amounts of labeled data.","url_abs":"https://arxiv.org/abs/2006.11477v3","url_pdf":"https://arxiv.org/pdf/2006.11477v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/pytorch/fairseq","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/AIdeaLab/wav2vec2_docker","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/Arizona-Voice/Arizona-spotting","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/BirgerMoell/tmh","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/HarunoriKawano/Wav2vec2.0","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/JoungheeKim/Non-Attentive-Tacotron","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/eastonYi/wav2vec","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/facebookresearch/brainmagick","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/gatech-eic/s3-router","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/huggingface/transformers","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/huseinzol05/malaya-speech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/liutianlin0121/seislm","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/mailong25/self-supervised-speech-recognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/mailong25/vietnamese-speech-recognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/neonbjb/ocotillo","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/nlp-en-es/wav2vec2-spanish","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"jax","reach":{"status":"ok"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/phanxuanphucnd/Arizona-asr","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/phanxuanphucnd/Arizona-spotting","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/phanxuanphucnd/wav2asr","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/sh-lee-prml/hierspeechpp","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/shivangi-aneja/FaceTalk","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/vasudevgupta7/gsoc-wav2vec2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/pwc-1/Paper-9/tree/main/1/wav2vec2_conformer","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"mindspore","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/pytorch/fairseq/tree/master/examples/wav2vec","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"wav2vec-2-0-a-framework-for-self-supervised","repo_url":"https://github.com/wenet-e2e/wenet","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"quantization","task_name":"Quantization"},{"task_slug":"self-supervised-learning","task_name":"Self-Supervised Learning"},{"task_slug":"speech-recognition","task_name":"Speech Recognition"},{"task_slug":"zero-shot-audio-retrieval","task_name":"Zero-Shot Audio Retrieval"}],"methods":[{"method_slug":"absolute-position-encodings","method_name":"Absolute Position Encodings"},{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"gumbel-softmax","method_name":"Gumbel Softmax"},{"method_slug":"label-smoothing","method_name":"Label Smoothing"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"position-wise-feed-forward-layer","method_name":"Position-Wise Feed-Forward Layer"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"transformer","method_name":"Transformer"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/speech-recognition-on-libri-light-test-clean","task":"Speech Recognition","dataset":"Libri-Light test-clean","model":"wav2vec 2.0 Large-10h-LV-60k","rank_in_archive_order":1,"of":5,"metrics":{"Word Error Rate (WER)":"2.5"},"uses_additional_data":false},{"leaderboard":"/sota/speech-recognition-on-libri-light-test-other","task":"Speech Recognition","dataset":"Libri-Light test-other","model":"wav2vec 2.0 Large-10h-LV-60k","rank_in_archive_order":1,"of":5,"metrics":{"Word Error Rate (WER)":"5.0"},"uses_additional_data":false},{"leaderboard":"/sota/speech-recognition-on-librispeech-test-clean","task":"Speech Recognition","dataset":"LibriSpeech test-clean","model":"wav2vec 2.0 with Libri-Light","rank_in_archive_order":12,"of":64,"metrics":{"Word Error Rate (WER)":"1.8"},"uses_additional_data":true},{"leaderboard":"/sota/speech-recognition-on-librispeech-test-other","task":"Speech Recognition","dataset":"LibriSpeech test-other","model":"wav2vec 2.0 with Libri-Light","rank_in_archive_order":6,"of":53,"metrics":{"Word Error Rate (WER)":"3.0"},"uses_additional_data":false},{"leaderboard":"/sota/speech-recognition-on-librispeech-test-other","task":"Speech Recognition","dataset":"LibriSpeech test-other","model":"wav2vec 2.0","rank_in_archive_order":17,"of":53,"metrics":{"Word Error Rate (WER)":"4.1"},"uses_additional_data":true},{"leaderboard":"/sota/speech-recognition-on-timit","task":"Speech Recognition","dataset":"TIMIT","model":"wav2vec 2.0","rank_in_archive_order":1,"of":22,"metrics":{"Percentage error":"8.3"},"uses_additional_data":true}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2006.11477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11477"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pytorch/fairseq","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huggingface/transformers","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/wenet-e2e/wenet","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/neonbjb/ocotillo","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/vasudevgupta7/gsoc-wav2vec2","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pwc-1/Paper-9/tree/main/1/wav2vec2_conformer","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/facebookresearch/brainmagick","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/liutianlin0121/seislm","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/eastonYi/wav2vec","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/gatech-eic/s3-router","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/HarunoriKawano/Wav2vec2.0","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/phanxuanphucnd/Arizona-spotting","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/nlp-en-es/wav2vec2-spanish","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/phanxuanphucnd/Arizona-asr","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/AIdeaLab/wav2vec2_docker","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mailong25/self-supervised-speech-recognition","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mailong25/vietnamese-speech-recognition","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Arizona-Voice/Arizona-spotting","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huseinzol05/malaya-speech","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/shivangi-aneja/FaceTalk","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/phanxuanphucnd/wav2asr","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/BirgerMoell/tmh","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pytorch/fairseq/tree/master/examples/wav2vec","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/JoungheeKim/Non-Attentive-Tacotron","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/sh-lee-prml/hierspeechpp","reach":null}],"summary":{"ran_draft_wrong":2,"unverified":7},"by_repo_kind":{"listed":{"samples":9,"ran":2,"repositories":3}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":2,"samples":[{"code_sha256_prefix":"0f3c0e3f4461c02d","entry":"add_asr_eval_argument","repo":"mailong25/self-supervised-speech-recognition","repo_kind":"listed","path":"stt.py","file_url":"https://github.com/mailong25/self-supervised-speech-recognition/blob/HEAD/stt.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0f3c0e3f4461c02d"}},{"code_sha256_prefix":"732cb84e2f814ba6","entry":"get_dataset_itr","repo":"mailong25/self-supervised-speech-recognition","repo_kind":"listed","path":"stt.py","file_url":"https://github.com/mailong25/self-supervised-speech-recognition/blob/HEAD/stt.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"732cb84e2f814ba6"}},{"code_sha256_prefix":"fa1c3918c88bf8f5","entry":"apply_spec_augmentation","repo":"vasudevgupta7/gsoc-wav2vec2","repo_kind":"listed","path":"src/wav2vec2/spec_augment.py","file_url":"https://github.com/vasudevgupta7/gsoc-wav2vec2/blob/HEAD/src/wav2vec2/spec_augment.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"fa1c3918c88bf8f5"}},{"code_sha256_prefix":"90190d4490581c2d","entry":"get_from_registry","repo":"phanxuanphucnd/wav2asr","repo_kind":"listed","path":"arizona/utils/misc_utils.py","file_url":"https://github.com/phanxuanphucnd/wav2asr/blob/HEAD/arizona/utils/misc_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"90190d4490581c2d"}},{"code_sha256_prefix":"ea6f42601b6cb44e","entry":"ifnone","repo":"phanxuanphucnd/wav2asr","repo_kind":"listed","path":"arizona/utils/misc_utils.py","file_url":"https://github.com/phanxuanphucnd/wav2asr/blob/HEAD/arizona/utils/misc_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ea6f42601b6cb44e"}},{"code_sha256_prefix":"d39dd4b7bcc000b8","entry":"read_tfrecords","repo":"vasudevgupta7/gsoc-wav2vec2","repo_kind":"listed","path":"src/data_utils.py","file_url":"https://github.com/vasudevgupta7/gsoc-wav2vec2/blob/HEAD/src/data_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d39dd4b7bcc000b8"}},{"code_sha256_prefix":"e488217cd68c52ed","entry":"replace","repo":"vasudevgupta7/gsoc-wav2vec2","repo_kind":"listed","path":"src/convert_torch_to_tf.py","file_url":"https://github.com/vasudevgupta7/gsoc-wav2vec2/blob/HEAD/src/convert_torch_to_tf.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"e488217cd68c52ed"}},{"code_sha256_prefix":"9e0b1e5fd4f77016","entry":"str2bool","repo":"phanxuanphucnd/wav2asr","repo_kind":"listed","path":"arizona/utils/misc_utils.py","file_url":"https://github.com/phanxuanphucnd/wav2asr/blob/HEAD/arizona/utils/misc_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"9e0b1e5fd4f77016"}},{"code_sha256_prefix":"d1e7bdb6b6c8d6cb","entry":"tf_multinomial_no_replacement","repo":"vasudevgupta7/gsoc-wav2vec2","repo_kind":"listed","path":"src/wav2vec2/spec_augment.py","file_url":"https://github.com/vasudevgupta7/gsoc-wav2vec2/blob/HEAD/src/wav2vec2/spec_augment.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d1e7bdb6b6c8d6cb"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}