{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/robust-vocal-quality-feature-embeddings-for","title":"Robust Vocal Quality Feature Embeddings for Dysphonic Voice Detection","arxiv_id":"2211.09858","date":"2022-11-17","proceeding":null,"authors":["Jianwei Zhang","Julie Liss","Suren Jayasuriya","Visar Berisha"],"abstract":"Approximately 1.2% of the world's population has impaired voice production. As a result, automatic dysphonic voice detection has attracted considerable academic and clinical interest. However, existing methods for automated voice assessment often fail to generalize outside the training conditions or to other related applications. In this paper, we propose a deep learning framework for generating acoustic feature embeddings sensitive to vocal quality and robust across different corpora. A contrastive loss is combined with a classification loss to train our deep learning model jointly. Data warping methods are used on input voice samples to improve the robustness of our method. Empirical results demonstrate that our method not only achieves high in-corpus and cross-corpus classification accuracy but also generates good embeddings sensitive to voice quality and robust across different corpora. We also compare our results against three baseline methods on clean and three variations of deteriorated in-corpus and cross-corpus datasets and demonstrate that the proposed model consistently outperforms the baseline methods.","url_abs":"https://arxiv.org/abs/2211.09858v2","url_pdf":"https://arxiv.org/pdf/2211.09858v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"robust-vocal-quality-feature-embeddings-for","repo_url":"https://github.com/vigor-jzhang/dysphonic-emb","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"cross-corpus","task_name":"Cross-corpus"}],"methods":[{"method_slug":"fail","method_name":"fail"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2211.09858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09858"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/vigor-jzhang/dysphonic-emb","reach":null}],"summary":{"ran_fixture":1,"ran_draft_wrong":2},"by_repo_kind":{"official":{"samples":3,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"89d938a5bbbf84f9","entry":"IR_aug","repo":"vigor-jzhang/dysphonic-emb","repo_kind":"official","path":"dataloader.py","file_url":"https://github.com/vigor-jzhang/dysphonic-emb/blob/HEAD/dataloader.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"89d938a5bbbf84f9"}},{"code_sha256_prefix":"b58fc8892dafff80","entry":"form_list","repo":"vigor-jzhang/dysphonic-emb","repo_kind":"official","path":"dataloader.py","file_url":"https://github.com/vigor-jzhang/dysphonic-emb/blob/HEAD/dataloader.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"b58fc8892dafff80"}},{"code_sha256_prefix":"62804c7a7d03c208","entry":"get_flist","repo":"vigor-jzhang/dysphonic-emb","repo_kind":"official","path":"dataloader.py","file_url":"https://github.com/vigor-jzhang/dysphonic-emb/blob/HEAD/dataloader.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"62804c7a7d03c208"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}