{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/speech-emotion-recognition-with-multi-task","title":"Speech Emotion Recognition with Multi-Task Learning","arxiv_id":null,"date":"2021-09-06","proceeding":"Interspeech 2021 9","authors":["Cai","Xingyu Yuan","Jiahong Zheng","Renjie Huang","Liang Church","Kenneth"],"abstract":"Speech emotion recognition (SER) classifies speech into emotion categories such as: Happy, Angry, Sad and Neutral. Recently , deep learning has been applied to the SER task. This paper proposes a multi-task learning (MTL) framework to simultaneously perform speech-to-text recognition and emotion classification, with an end-to-end deep neural model based on wav2vec-2.0. Experiments on the IEMOCAP benchmark show that the proposed method achieves the state-of-the-art performance on the SER task. In addition, an ablation study establishes the effectiveness of the proposed MTL framework.","url_abs":"https://www.isca-speech.org/archive/interspeech_2021/cai21b_interspeech.html","url_pdf":"https://www.isca-speech.org/archive/interspeech_2021/cai21b_interspeech.html","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"speech-emotion-recognition-with-multi-task","repo_url":"https://github.com/TideDancer/interspeech21_emotion","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"emotion-classification","task_name":"Emotion Classification"},{"task_slug":"emotion-recognition","task_name":"Emotion Recognition"},{"task_slug":"multi-task-learning","task_name":"Multi-Task Learning"},{"task_slug":"speech-emotion-recognition","task_name":"Speech Emotion Recognition"},{"task_slug":"speech-to-text","task_name":"Speech-to-Text"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/speech-emotion-recognition-on-iemocap","task":"Speech Emotion Recognition","dataset":"IEMOCAP","model":"SER with MTL","rank_in_archive_order":1,"of":8,"metrics":{"F1":"-","UA CV":"0.7815"},"uses_additional_data":false}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}