{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/speechocean762-an-open-source-non-native","title":"speechocean762: An Open-Source Non-native English Speech Corpus For Pronunciation Assessment","arxiv_id":"2104.01378","date":"2021-04-03","proceeding":null,"authors":["Junbo Zhang","Zhiwen Zhang","Yongqing Wang","Zhiyong Yan","Qiong Song","YuKai Huang","Ke Li","Daniel Povey","Yujun Wang"],"abstract":"This paper introduces a new open-source speech corpus named \"speechocean762\" designed for pronunciation assessment use, consisting of 5000 English utterances from 250 non-native speakers, where half of the speakers are children. Five experts annotated each of the utterances at sentence-level, word-level and phoneme-level. A baseline system is released in open source to illustrate the phoneme-level pronunciation assessment workflow on this corpus. This corpus is allowed to be used freely for commercial and non-commercial purposes. It is available for free download from OpenSLR, and the corresponding baseline system is published in the Kaldi speech recognition toolkit.","url_abs":"https://arxiv.org/abs/2104.01378v2","url_pdf":"https://arxiv.org/pdf/2104.01378v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"speechocean762-an-open-source-non-native","repo_url":"https://github.com/kaldi-asr/kaldi/tree/master/egs/gop_speechocean762","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":null},{"paper_slug":"speechocean762-an-open-source-non-native","repo_url":"https://github.com/YuanGongND/gopt","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"phone-level-pronunciation-scoring","task_name":"Phone-level pronunciation scoring"},{"task_slug":"sentence","task_name":"Sentence"},{"task_slug":"speech-recognition-1","task_name":"speech-recognition"}],"methods":[],"datasets_introduced":[{"slug":"speechocean762","name":"speechocean762","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/phone-level-pronunciation-scoring-on","task":"Phone-level pronunciation scoring","dataset":"speechocean762","model":"GOP","rank_in_archive_order":8,"of":8,"metrics":{"Pearson correlation coefficient (PCC)":"0.45"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2104.01378","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}